{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:VJ5XRZEZFRCBYJFKRZO5QDPAUP","short_pith_number":"pith:VJ5XRZEZ","schema_version":"1.0","canonical_sha256":"aa7b78e4992c441c24aa8e5dd80de0a3dd5f7fb525a79c9aa277534b28360a3a","source":{"kind":"arxiv","id":"2005.00547","version":2},"attestation_state":"computed","paper":{"title":"GoEmotions: A Dataset of Fine-Grained Emotions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alan Cowen, Dana Movshovitz-Attias, Dorottya Demszky, Gaurav Nemade, Jeongwoo Ko, Sujith Ravi","submitted_at":"2020-05-01T18:00:02Z","abstract_excerpt":"Understanding emotion expressed in language has a wide range of applications, from building empathetic chatbots to detecting harmful online behavior. Advancement in this area can be improved using large-scale datasets with a fine-grained typology, adaptable to multiple downstream tasks. We introduce GoEmotions, the largest manually annotated dataset of 58k English Reddit comments, labeled for 27 emotion categories or Neutral. We demonstrate the high quality of the annotations via Principal Preserved Component Analysis. We conduct transfer learning experiments with existing emotion benchmarks t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.00547","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-05-01T18:00:02Z","cross_cats_sorted":[],"title_canon_sha256":"dc955d6fead70bb18861715d4ee750c6dfb00c45641b12dce86df8699d307d9d","abstract_canon_sha256":"1dd4bd05edc9352d13cd6a709c0b90410699b97e713410b4602cfb82a96fc513"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:07:41.006658Z","signature_b64":"4unWkQoc4WvMOC3FuFtQd56zteuzFhqwohhNMggMTkkiUtl0j5sDpMFN3+rMia19KusKyDKc8yzw680+72MNDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa7b78e4992c441c24aa8e5dd80de0a3dd5f7fb525a79c9aa277534b28360a3a","last_reissued_at":"2026-07-05T01:07:41.006156Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:07:41.006156Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GoEmotions: A Dataset of Fine-Grained Emotions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alan Cowen, Dana Movshovitz-Attias, Dorottya Demszky, Gaurav Nemade, Jeongwoo Ko, Sujith Ravi","submitted_at":"2020-05-01T18:00:02Z","abstract_excerpt":"Understanding emotion expressed in language has a wide range of applications, from building empathetic chatbots to detecting harmful online behavior. Advancement in this area can be improved using large-scale datasets with a fine-grained typology, adaptable to multiple downstream tasks. We introduce GoEmotions, the largest manually annotated dataset of 58k English Reddit comments, labeled for 27 emotion categories or Neutral. We demonstrate the high quality of the annotations via Principal Preserved Component Analysis. We conduct transfer learning experiments with existing emotion benchmarks t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.00547","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.00547/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.00547","created_at":"2026-07-05T01:07:41.006219+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.00547v2","created_at":"2026-07-05T01:07:41.006219+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.00547","created_at":"2026-07-05T01:07:41.006219+00:00"},{"alias_kind":"pith_short_12","alias_value":"VJ5XRZEZFRCB","created_at":"2026-07-05T01:07:41.006219+00:00"},{"alias_kind":"pith_short_16","alias_value":"VJ5XRZEZFRCBYJFK","created_at":"2026-07-05T01:07:41.006219+00:00"},{"alias_kind":"pith_short_8","alias_value":"VJ5XRZEZ","created_at":"2026-07-05T01:07:41.006219+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06875","citing_title":"Video2Reaction: Mapping Video to Audience Reaction Distribution in the Wild","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00046","citing_title":"When Jokes Cross the Line: Analyzing Regular Humor and Dark Humor in YouTube Shorts","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29273","citing_title":"A Hybrid Framework for Song Lyric Annotation Based on Human-LLM Alignment","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04152","citing_title":"SocialLM: Social Signal Processing of Patient-Provider Communication using LLMs and Contextual Aggregation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04070","citing_title":"T-FIX: Text-Based Explanations with Features Interpretable to eXperts","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13674","citing_title":"PrefixMemory-Tuning: Modernizing Prefix-Tuning by Decoupling the Prefix from Attention","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04947","citing_title":"SUMMIR: A Hallucination-Aware Framework for Ranking Sports Insights from LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02655","citing_title":"Semantic Data Processing with Holistic Data Understanding","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28052","citing_title":"Meta-Harness: End-to-End Optimization of Model Harnesses","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23719","citing_title":"AIPsy-Affect: A Keyword-Free Clinical Stimulus Battery for Mechanistic Interpretability of Emotion in Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP","json":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP.json","graph_json":"https://pith.science/api/pith-number/VJ5XRZEZFRCBYJFKRZO5QDPAUP/graph.json","events_json":"https://pith.science/api/pith-number/VJ5XRZEZFRCBYJFKRZO5QDPAUP/events.json","paper":"https://pith.science/paper/VJ5XRZEZ"},"agent_actions":{"view_html":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP","download_json":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP.json","view_paper":"https://pith.science/paper/VJ5XRZEZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.00547&json=true","fetch_graph":"https://pith.science/api/pith-number/VJ5XRZEZFRCBYJFKRZO5QDPAUP/graph.json","fetch_events":"https://pith.science/api/pith-number/VJ5XRZEZFRCBYJFKRZO5QDPAUP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP/action/storage_attestation","attest_author":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP/action/author_attestation","sign_citation":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP/action/citation_signature","submit_replication":"https://pith.science/pith/VJ5XRZEZFRCBYJFKRZO5QDPAUP/action/replication_record"}},"created_at":"2026-07-05T01:07:41.006219+00:00","updated_at":"2026-07-05T01:07:41.006219+00:00"}