{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QWXQ5Z44GGHS4XBOKPMD3WMU3I","short_pith_number":"pith:QWXQ5Z44","schema_version":"1.0","canonical_sha256":"85af0ee79c318f2e5c2e53d83dd994da2387206f572ceb12b307332faa0e8bf0","source":{"kind":"arxiv","id":"2407.11300","version":1},"attestation_state":"computed","paper":{"title":"Large Vision-Language Models as Emotion Recognizers in Context Awareness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Dingkang Yang, Jiawei Chen, Lihua Zhang, Peng Zhai, Yuxuan Lei, Zhaoyu Chen","submitted_at":"2024-07-16T01:28:06Z","abstract_excerpt":"Context-aware emotion recognition (CAER) is a complex and significant task that requires perceiving emotions from various contextual cues. Previous approaches primarily focus on designing sophisticated architectures to extract emotional cues from images. However, their knowledge is confined to specific training datasets and may reflect the subjective emotional biases of the annotators. Furthermore, acquiring large amounts of labeled data is often challenging in real-world applications. In this paper, we systematically explore the potential of leveraging Large Vision-Language Models (LVLMs) to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.11300","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-16T01:28:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"522d38717c2981def7cb6b3d51e70d312a917b858fdab49b39624f36c41e3f7c","abstract_canon_sha256":"1e44cb0ab4ca5e1cac2c72a15b2e13aec4b670ef5dc47669f815db78fd027d1d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:21.866082Z","signature_b64":"ROzd46U0Grr85P6wpvcYO/xyf63V5i0svo09mR+9H75C8nuBjfs1iOHBhkJNCU3ptm3XTfiw2nXrJMkA5OLfAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85af0ee79c318f2e5c2e53d83dd994da2387206f572ceb12b307332faa0e8bf0","last_reissued_at":"2026-07-05T08:44:21.865547Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:21.865547Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Vision-Language Models as Emotion Recognizers in Context Awareness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Dingkang Yang, Jiawei Chen, Lihua Zhang, Peng Zhai, Yuxuan Lei, Zhaoyu Chen","submitted_at":"2024-07-16T01:28:06Z","abstract_excerpt":"Context-aware emotion recognition (CAER) is a complex and significant task that requires perceiving emotions from various contextual cues. Previous approaches primarily focus on designing sophisticated architectures to extract emotional cues from images. However, their knowledge is confined to specific training datasets and may reflect the subjective emotional biases of the annotators. Furthermore, acquiring large amounts of labeled data is often challenging in real-world applications. In this paper, we systematically explore the potential of leveraging Large Vision-Language Models (LVLMs) to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.11300","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.11300/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.11300","created_at":"2026-07-05T08:44:21.865613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.11300v1","created_at":"2026-07-05T08:44:21.865613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.11300","created_at":"2026-07-05T08:44:21.865613+00:00"},{"alias_kind":"pith_short_12","alias_value":"QWXQ5Z44GGHS","created_at":"2026-07-05T08:44:21.865613+00:00"},{"alias_kind":"pith_short_16","alias_value":"QWXQ5Z44GGHS4XBO","created_at":"2026-07-05T08:44:21.865613+00:00"},{"alias_kind":"pith_short_8","alias_value":"QWXQ5Z44","created_at":"2026-07-05T08:44:21.865613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25325","citing_title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22072","citing_title":"A Controlled Study of CLIP-Based Body-Scene Fusion for Emotion Recognition in Context","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12605","citing_title":"Multimodal Chain-of-Thought Reasoning: A Comprehensive Survey","ref_index":219,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I","json":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I.json","graph_json":"https://pith.science/api/pith-number/QWXQ5Z44GGHS4XBOKPMD3WMU3I/graph.json","events_json":"https://pith.science/api/pith-number/QWXQ5Z44GGHS4XBOKPMD3WMU3I/events.json","paper":"https://pith.science/paper/QWXQ5Z44"},"agent_actions":{"view_html":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I","download_json":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I.json","view_paper":"https://pith.science/paper/QWXQ5Z44","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.11300&json=true","fetch_graph":"https://pith.science/api/pith-number/QWXQ5Z44GGHS4XBOKPMD3WMU3I/graph.json","fetch_events":"https://pith.science/api/pith-number/QWXQ5Z44GGHS4XBOKPMD3WMU3I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I/action/storage_attestation","attest_author":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I/action/author_attestation","sign_citation":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I/action/citation_signature","submit_replication":"https://pith.science/pith/QWXQ5Z44GGHS4XBOKPMD3WMU3I/action/replication_record"}},"created_at":"2026-07-05T08:44:21.865613+00:00","updated_at":"2026-07-05T08:44:21.865613+00:00"}