{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HSKD4CWMD4F54OQY76Q54X2Y5H","short_pith_number":"pith:HSKD4CWM","schema_version":"1.0","canonical_sha256":"3c943e0acc1f0bde3a18ffa1de5f58e9fed49c03eac4f9e4cd950a7871257746","source":{"kind":"arxiv","id":"2411.17672","version":1},"attestation_state":"computed","paper":{"title":"Synthetic Data Generation with LLM for Improved Depression Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrea Kang, Jun Yu Chen, Shuhao Fu, Zoe Lee-Youngzie","submitted_at":"2024-11-26T18:31:14Z","abstract_excerpt":"Automatic detection of depression is a rapidly growing field of research at the intersection of psychology and machine learning. However, with its exponential interest comes a growing concern for data privacy and scarcity due to the sensitivity of such a topic. In this paper, we propose a pipeline for Large Language Models (LLMs) to generate synthetic data to improve the performance of depression prediction models. Starting from unstructured, naturalistic text data from recorded transcripts of clinical interviews, we utilize an open-source LLM to generate synthetic data through chain-of-though"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17672","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-26T18:31:14Z","cross_cats_sorted":[],"title_canon_sha256":"e33b0d0e10aeb00983950dad28afb02d38232951bd7f92d0f933e90b4b7bfd9f","abstract_canon_sha256":"86f4958c8dcd63971f90ab22dbdfcc6b5e845271216eb446a265b860135b440f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:40:50.286978Z","signature_b64":"qQy/IM28NMA9tk+JXXazqovvvVAzhBZRFOFPNvCjsqKblWmiOU6zlMTfnGabcOSw2jdKLdPIRyJOXywwJVtOCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c943e0acc1f0bde3a18ffa1de5f58e9fed49c03eac4f9e4cd950a7871257746","last_reissued_at":"2026-07-05T09:40:50.286419Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:40:50.286419Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Synthetic Data Generation with LLM for Improved Depression Prediction","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrea Kang, Jun Yu Chen, Shuhao Fu, Zoe Lee-Youngzie","submitted_at":"2024-11-26T18:31:14Z","abstract_excerpt":"Automatic detection of depression is a rapidly growing field of research at the intersection of psychology and machine learning. However, with its exponential interest comes a growing concern for data privacy and scarcity due to the sensitivity of such a topic. In this paper, we propose a pipeline for Large Language Models (LLMs) to generate synthetic data to improve the performance of depression prediction models. Starting from unstructured, naturalistic text data from recorded transcripts of clinical interviews, we utilize an open-source LLM to generate synthetic data through chain-of-though"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17672","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17672/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17672","created_at":"2026-07-05T09:40:50.286494+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17672v1","created_at":"2026-07-05T09:40:50.286494+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17672","created_at":"2026-07-05T09:40:50.286494+00:00"},{"alias_kind":"pith_short_12","alias_value":"HSKD4CWMD4F5","created_at":"2026-07-05T09:40:50.286494+00:00"},{"alias_kind":"pith_short_16","alias_value":"HSKD4CWMD4F54OQY","created_at":"2026-07-05T09:40:50.286494+00:00"},{"alias_kind":"pith_short_8","alias_value":"HSKD4CWM","created_at":"2026-07-05T09:40:50.286494+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19640","citing_title":"Creating Multilingual Mental Health Dialogue Datasets: Limits of Persona-Based Localization via Nationality and Language","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13318","citing_title":"VERA-MH: Validation of Ethical and Responsible AI in Mental Health","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13318","citing_title":"VERA-MH: Validation of Ethical and Responsible AI in Mental Health","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05752","citing_title":"Generative AI-Based Monte Carlo Simulation for Method Evaluation Using Synthetic Multilevel Data","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H","json":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H.json","graph_json":"https://pith.science/api/pith-number/HSKD4CWMD4F54OQY76Q54X2Y5H/graph.json","events_json":"https://pith.science/api/pith-number/HSKD4CWMD4F54OQY76Q54X2Y5H/events.json","paper":"https://pith.science/paper/HSKD4CWM"},"agent_actions":{"view_html":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H","download_json":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H.json","view_paper":"https://pith.science/paper/HSKD4CWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17672&json=true","fetch_graph":"https://pith.science/api/pith-number/HSKD4CWMD4F54OQY76Q54X2Y5H/graph.json","fetch_events":"https://pith.science/api/pith-number/HSKD4CWMD4F54OQY76Q54X2Y5H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H/action/storage_attestation","attest_author":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H/action/author_attestation","sign_citation":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H/action/citation_signature","submit_replication":"https://pith.science/pith/HSKD4CWMD4F54OQY76Q54X2Y5H/action/replication_record"}},"created_at":"2026-07-05T09:40:50.286494+00:00","updated_at":"2026-07-05T09:40:50.286494+00:00"}