{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6LDXS4D47FIJTTQQFBIKGHKIIR","short_pith_number":"pith:6LDXS4D4","schema_version":"1.0","canonical_sha256":"f2c779707cf95099ce102850a31d48445d34f494b7279d07a724aa4a36c81364","source":{"kind":"arxiv","id":"2104.09644","version":1},"attestation_state":"computed","paper":{"title":"Neural Language Models with Distant Supervision to Identify Major Depressive Disorder from Clinical Notes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Bhavani Singh Agnikula Kshatriya, Brandon J Coombes, Euijung Ryu, Joanna M Biernacka, Manuel Gardea- Resendez, Mark A Frye, Nicolas A Nunez, Sunyang Fu, Yanshan Wang","submitted_at":"2021-04-19T21:11:41Z","abstract_excerpt":"Major depressive disorder (MDD) is a prevalent psychiatric disorder that is associated with significant healthcare burden worldwide. Phenotyping of MDD can help early diagnosis and consequently may have significant advantages in patient management. In prior research MDD phenotypes have been extracted from structured Electronic Health Records (EHR) or using Electroencephalographic (EEG) data with traditional machine learning models to predict MDD phenotypes. However, MDD phenotypic information is also documented in free-text EHR data, such as clinical notes. While clinical notes may provide mor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.09644","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-04-19T21:11:41Z","cross_cats_sorted":["cs.AI","cs.IR"],"title_canon_sha256":"394da77b82cb6a75ec18c69818e73cf53760e326db2eb4aae7be18740afa468e","abstract_canon_sha256":"461cb6ac883cb782bcc3cd1954100851d1a783f13391b60dbc8d33fd9b1e3319"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:33:35.612995Z","signature_b64":"rvfMec44wovBPP0JN9ydZSptnsvGJoAIirComxWk9m7VowVHMxU0ew6G4Bc1WbTEtfpQ2/fwfQO1ZX17ApiAAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f2c779707cf95099ce102850a31d48445d34f494b7279d07a724aa4a36c81364","last_reissued_at":"2026-07-05T02:33:35.612458Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:33:35.612458Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neural Language Models with Distant Supervision to Identify Major Depressive Disorder from Clinical Notes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR"],"primary_cat":"cs.CL","authors_text":"Bhavani Singh Agnikula Kshatriya, Brandon J Coombes, Euijung Ryu, Joanna M Biernacka, Manuel Gardea- Resendez, Mark A Frye, Nicolas A Nunez, Sunyang Fu, Yanshan Wang","submitted_at":"2021-04-19T21:11:41Z","abstract_excerpt":"Major depressive disorder (MDD) is a prevalent psychiatric disorder that is associated with significant healthcare burden worldwide. Phenotyping of MDD can help early diagnosis and consequently may have significant advantages in patient management. In prior research MDD phenotypes have been extracted from structured Electronic Health Records (EHR) or using Electroencephalographic (EEG) data with traditional machine learning models to predict MDD phenotypes. However, MDD phenotypic information is also documented in free-text EHR data, such as clinical notes. While clinical notes may provide mor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.09644","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.09644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.09644","created_at":"2026-07-05T02:33:35.612504+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.09644v1","created_at":"2026-07-05T02:33:35.612504+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.09644","created_at":"2026-07-05T02:33:35.612504+00:00"},{"alias_kind":"pith_short_12","alias_value":"6LDXS4D47FIJ","created_at":"2026-07-05T02:33:35.612504+00:00"},{"alias_kind":"pith_short_16","alias_value":"6LDXS4D47FIJTTQQ","created_at":"2026-07-05T02:33:35.612504+00:00"},{"alias_kind":"pith_short_8","alias_value":"6LDXS4D4","created_at":"2026-07-05T02:33:35.612504+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.02848","citing_title":"Aligning Large Language Models with Healthcare Stakeholders: A Pathway to Trustworthy AI Integration","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR","json":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR.json","graph_json":"https://pith.science/api/pith-number/6LDXS4D47FIJTTQQFBIKGHKIIR/graph.json","events_json":"https://pith.science/api/pith-number/6LDXS4D47FIJTTQQFBIKGHKIIR/events.json","paper":"https://pith.science/paper/6LDXS4D4"},"agent_actions":{"view_html":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR","download_json":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR.json","view_paper":"https://pith.science/paper/6LDXS4D4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.09644&json=true","fetch_graph":"https://pith.science/api/pith-number/6LDXS4D47FIJTTQQFBIKGHKIIR/graph.json","fetch_events":"https://pith.science/api/pith-number/6LDXS4D47FIJTTQQFBIKGHKIIR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR/action/storage_attestation","attest_author":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR/action/author_attestation","sign_citation":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR/action/citation_signature","submit_replication":"https://pith.science/pith/6LDXS4D47FIJTTQQFBIKGHKIIR/action/replication_record"}},"created_at":"2026-07-05T02:33:35.612504+00:00","updated_at":"2026-07-05T02:33:35.612504+00:00"}