{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DIB27L4MCJ6N4BPXPBXRB5YJFJ","short_pith_number":"pith:DIB27L4M","schema_version":"1.0","canonical_sha256":"1a03afaf8c127cde05f7786f10f7092a6236bc4482296394a7f10d7c0443f0e7","source":{"kind":"arxiv","id":"2502.05660","version":1},"attestation_state":"computed","paper":{"title":"Evaluating Vision-Language Models for Emotion Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"James Z. Wang, Sree Bhattacharyya","submitted_at":"2025-02-08T18:25:31Z","abstract_excerpt":"Large Vision-Language Models (VLMs) have achieved unprecedented success in several objective multimodal reasoning tasks. However, to further enhance their capabilities of empathetic and effective communication with humans, improving how VLMs process and understand emotions is crucial. Despite significant research attention on improving affective understanding, there is a lack of detailed evaluations of VLMs for emotion-related tasks, which can potentially help inform downstream fine-tuning efforts. In this work, we present the first comprehensive evaluation of VLMs for recognizing evoked emoti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05660","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-08T18:25:31Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"dc72778e4a90f7f5406b43d47c359f9318952de17d8d7b80c5250e33304da7b0","abstract_canon_sha256":"a8fff5b93635c54226e647a82bd984e27b5a4a1903fe27bf58df90d274845736"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:41.691266Z","signature_b64":"OW68Gea0YHfu9IbaJ7Q4f1yufwYX+etpdXccOLBPZNrHoPuEWkkgANQ8YUTZ21JFdCT+aEB58fbWGwgPt8ZODg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a03afaf8c127cde05f7786f10f7092a6236bc4482296394a7f10d7c0443f0e7","last_reissued_at":"2026-07-05T10:11:41.690853Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:41.690853Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Vision-Language Models for Emotion Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"James Z. Wang, Sree Bhattacharyya","submitted_at":"2025-02-08T18:25:31Z","abstract_excerpt":"Large Vision-Language Models (VLMs) have achieved unprecedented success in several objective multimodal reasoning tasks. However, to further enhance their capabilities of empathetic and effective communication with humans, improving how VLMs process and understand emotions is crucial. Despite significant research attention on improving affective understanding, there is a lack of detailed evaluations of VLMs for emotion-related tasks, which can potentially help inform downstream fine-tuning efforts. In this work, we present the first comprehensive evaluation of VLMs for recognizing evoked emoti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05660","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05660/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05660","created_at":"2026-07-05T10:11:41.690904+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05660v1","created_at":"2026-07-05T10:11:41.690904+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05660","created_at":"2026-07-05T10:11:41.690904+00:00"},{"alias_kind":"pith_short_12","alias_value":"DIB27L4MCJ6N","created_at":"2026-07-05T10:11:41.690904+00:00"},{"alias_kind":"pith_short_16","alias_value":"DIB27L4MCJ6N4BPX","created_at":"2026-07-05T10:11:41.690904+00:00"},{"alias_kind":"pith_short_8","alias_value":"DIB27L4M","created_at":"2026-07-05T10:11:41.690904+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02089","citing_title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12554","citing_title":"EmoVerse: A MLLMs-Driven Emotion Representation Dataset for Interpretable Visual Emotion Analysis","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ","json":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ.json","graph_json":"https://pith.science/api/pith-number/DIB27L4MCJ6N4BPXPBXRB5YJFJ/graph.json","events_json":"https://pith.science/api/pith-number/DIB27L4MCJ6N4BPXPBXRB5YJFJ/events.json","paper":"https://pith.science/paper/DIB27L4M"},"agent_actions":{"view_html":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ","download_json":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ.json","view_paper":"https://pith.science/paper/DIB27L4M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05660&json=true","fetch_graph":"https://pith.science/api/pith-number/DIB27L4MCJ6N4BPXPBXRB5YJFJ/graph.json","fetch_events":"https://pith.science/api/pith-number/DIB27L4MCJ6N4BPXPBXRB5YJFJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ/action/storage_attestation","attest_author":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ/action/author_attestation","sign_citation":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ/action/citation_signature","submit_replication":"https://pith.science/pith/DIB27L4MCJ6N4BPXPBXRB5YJFJ/action/replication_record"}},"created_at":"2026-07-05T10:11:41.690904+00:00","updated_at":"2026-07-05T10:11:41.690904+00:00"}