{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WI5OOPAVYW73V42VBG62CZQ2KN","short_pith_number":"pith:WI5OOPAV","schema_version":"1.0","canonical_sha256":"b23ae73c15c5bfbaf35509bda1661a53406ad48a070926b446f0a86c0788512d","source":{"kind":"arxiv","id":"2406.16442","version":2},"attestation_state":"computed","paper":{"title":"EmoLLM: Multimodal Emotional Understanding Meets Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Du, Mang Ye, Qu Yang","submitted_at":"2024-06-24T08:33:02Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have achieved remarkable performance on objective multimodal perception tasks, but their ability to interpret subjective, emotionally nuanced multimodal content remains largely unexplored. Thus, it impedes their ability to effectively understand and react to the intricate emotions expressed by humans through multimodal media. To bridge this gap, we introduce EmoBench, the first comprehensive benchmark designed specifically to evaluate the emotional capabilities of MLLMs across five popular emotional tasks, using a diverse dataset of 287k images and vid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16442","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T08:33:02Z","cross_cats_sorted":[],"title_canon_sha256":"936e009420b78ca0e2a28cc4e0c70887e919d9ec50a3227adc7221b893795537","abstract_canon_sha256":"0c4439e6a8b362f6055d06867ff2b034319cc9edf7726a9aea91e3137bb28d98"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:11.732065Z","signature_b64":"JcgInMmqk3hVXjASjRzsFIo7F5DT4JxS4SJ34xD/2CGnvnLBYNZS7eyofc0iiN5W8YRnQpalLilEXXXjY8WOCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b23ae73c15c5bfbaf35509bda1661a53406ad48a070926b446f0a86c0788512d","last_reissued_at":"2026-07-05T08:38:11.731559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:11.731559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EmoLLM: Multimodal Emotional Understanding Meets Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Du, Mang Ye, Qu Yang","submitted_at":"2024-06-24T08:33:02Z","abstract_excerpt":"Multi-modal large language models (MLLMs) have achieved remarkable performance on objective multimodal perception tasks, but their ability to interpret subjective, emotionally nuanced multimodal content remains largely unexplored. Thus, it impedes their ability to effectively understand and react to the intricate emotions expressed by humans through multimodal media. To bridge this gap, we introduce EmoBench, the first comprehensive benchmark designed specifically to evaluate the emotional capabilities of MLLMs across five popular emotional tasks, using a diverse dataset of 287k images and vid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16442","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16442/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16442","created_at":"2026-07-05T08:38:11.731619+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16442v2","created_at":"2026-07-05T08:38:11.731619+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16442","created_at":"2026-07-05T08:38:11.731619+00:00"},{"alias_kind":"pith_short_12","alias_value":"WI5OOPAVYW73","created_at":"2026-07-05T08:38:11.731619+00:00"},{"alias_kind":"pith_short_16","alias_value":"WI5OOPAVYW73V42V","created_at":"2026-07-05T08:38:11.731619+00:00"},{"alias_kind":"pith_short_8","alias_value":"WI5OOPAV","created_at":"2026-07-05T08:38:11.731619+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18988","citing_title":"ThinkDeception: A Progressive Reinforcement Learning Framework for Interpretable Multimodal Deception Detection","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02171","citing_title":"InsightVQA: High-Dimensional Emotion-Cognitive Visual Question Answering Benchmark","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2502.20689","citing_title":"WiseMind: a knowledge-guided multi-agent framework for accurate and empathetic psychiatric diagnosis","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18884","citing_title":"Navigating the Emotion Tree: Hierarchical Hyperbolic RAG for Multimodal Emotion Recognition","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09082","citing_title":"AVA-Bench: Atomic Visual Ability Benchmark for Vision Foundation Models","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02123","citing_title":"Nano-EmoX: Unifying Multimodal Emotional Intelligence from Perception to Empathy","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12735","citing_title":"AffectAgent: Collaborative Multi-Agent Reasoning for Retrieval-Augmented Multimodal Emotion Recognition","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08990","citing_title":"ActFER: Agentic Facial Expression Recognition via Active Tool-Augmented Visual Reasoning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05900","citing_title":"AICA-Bench: Holistically Examining the Capabilities of VLMs in Affective Image Content Analysis","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN","json":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN.json","graph_json":"https://pith.science/api/pith-number/WI5OOPAVYW73V42VBG62CZQ2KN/graph.json","events_json":"https://pith.science/api/pith-number/WI5OOPAVYW73V42VBG62CZQ2KN/events.json","paper":"https://pith.science/paper/WI5OOPAV"},"agent_actions":{"view_html":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN","download_json":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN.json","view_paper":"https://pith.science/paper/WI5OOPAV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16442&json=true","fetch_graph":"https://pith.science/api/pith-number/WI5OOPAVYW73V42VBG62CZQ2KN/graph.json","fetch_events":"https://pith.science/api/pith-number/WI5OOPAVYW73V42VBG62CZQ2KN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN/action/storage_attestation","attest_author":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN/action/author_attestation","sign_citation":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN/action/citation_signature","submit_replication":"https://pith.science/pith/WI5OOPAVYW73V42VBG62CZQ2KN/action/replication_record"}},"created_at":"2026-07-05T08:38:11.731619+00:00","updated_at":"2026-07-05T08:38:11.731619+00:00"}