{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JCCYLYVO2GXABSFOV2JFKRLPB2","short_pith_number":"pith:JCCYLYVO","schema_version":"1.0","canonical_sha256":"488585e2aed1ae00c8aeae9255456f0e8a52667f7e5a366384f2181fdc51c581","source":{"kind":"arxiv","id":"2408.11424","version":1},"attestation_state":"computed","paper":{"title":"EMO-LLaMA: Enhancing Facial Emotion Understanding with Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohao Xing, Heikki K\\\"alvi\\\"ainen, Huanjing Yue, Jingyu Yang, Kaishen Yuan, Qilang Ye, Weicheng Xie, Xin Liu, Zitong Yu","submitted_at":"2024-08-21T08:28:40Z","abstract_excerpt":"Facial expression recognition (FER) is an important research topic in emotional artificial intelligence. In recent decades, researchers have made remarkable progress. However, current FER paradigms face challenges in generalization, lack semantic information aligned with natural language, and struggle to process both images and videos within a unified framework, making their application in multimodal emotion understanding and human-computer interaction difficult. Multimodal Large Language Models (MLLMs) have recently achieved success, offering advantages in addressing these issues and potentia"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11424","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-21T08:28:40Z","cross_cats_sorted":[],"title_canon_sha256":"478a0235b5dcba6b9545f4ddedb7f8a4cf55e0e975cabbdbd120c41221e2edfc","abstract_canon_sha256":"44e4872d5941bb6e6fb178f41c7372ddc89a1ccb4c6e979d34df4237ee033f17"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:57:43.859344Z","signature_b64":"mPRqd2DEd7nMJLjabCHgyTqdcJzYTgqxlLv+8tUzp4jeQabYB2YuPFfEWPu/8xuYUGpeq5e6iq2uHjg7JEiCCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"488585e2aed1ae00c8aeae9255456f0e8a52667f7e5a366384f2181fdc51c581","last_reissued_at":"2026-07-05T08:57:43.858864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:57:43.858864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EMO-LLaMA: Enhancing Facial Emotion Understanding with Instruction Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohao Xing, Heikki K\\\"alvi\\\"ainen, Huanjing Yue, Jingyu Yang, Kaishen Yuan, Qilang Ye, Weicheng Xie, Xin Liu, Zitong Yu","submitted_at":"2024-08-21T08:28:40Z","abstract_excerpt":"Facial expression recognition (FER) is an important research topic in emotional artificial intelligence. In recent decades, researchers have made remarkable progress. However, current FER paradigms face challenges in generalization, lack semantic information aligned with natural language, and struggle to process both images and videos within a unified framework, making their application in multimodal emotion understanding and human-computer interaction difficult. Multimodal Large Language Models (MLLMs) have recently achieved success, offering advantages in addressing these issues and potentia"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11424","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11424","created_at":"2026-07-05T08:57:43.858922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11424v1","created_at":"2026-07-05T08:57:43.858922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11424","created_at":"2026-07-05T08:57:43.858922+00:00"},{"alias_kind":"pith_short_12","alias_value":"JCCYLYVO2GXA","created_at":"2026-07-05T08:57:43.858922+00:00"},{"alias_kind":"pith_short_16","alias_value":"JCCYLYVO2GXABSFO","created_at":"2026-07-05T08:57:43.858922+00:00"},{"alias_kind":"pith_short_8","alias_value":"JCCYLYVO","created_at":"2026-07-05T08:57:43.858922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08612","citing_title":"Facial Expression Recognition in the Deep Learning Era: A Systematic Multi-Criteria Review of Methods, Models, Datasets, Performance, Challenges, and Future Research Directions","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18073","citing_title":"FPBench: A Comprehensive Benchmark of Multimodal Large Language Models for Fingerprint Analysis","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08990","citing_title":"ActFER: Agentic Facial Expression Recognition via Active Tool-Augmented Visual Reasoning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06783","citing_title":"Insights from Visual Cognition: Understanding Human Action Dynamics with Overall Glance and Refined Gaze Transformer","ref_index":93,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2","json":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2.json","graph_json":"https://pith.science/api/pith-number/JCCYLYVO2GXABSFOV2JFKRLPB2/graph.json","events_json":"https://pith.science/api/pith-number/JCCYLYVO2GXABSFOV2JFKRLPB2/events.json","paper":"https://pith.science/paper/JCCYLYVO"},"agent_actions":{"view_html":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2","download_json":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2.json","view_paper":"https://pith.science/paper/JCCYLYVO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11424&json=true","fetch_graph":"https://pith.science/api/pith-number/JCCYLYVO2GXABSFOV2JFKRLPB2/graph.json","fetch_events":"https://pith.science/api/pith-number/JCCYLYVO2GXABSFOV2JFKRLPB2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2/action/storage_attestation","attest_author":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2/action/author_attestation","sign_citation":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2/action/citation_signature","submit_replication":"https://pith.science/pith/JCCYLYVO2GXABSFOV2JFKRLPB2/action/replication_record"}},"created_at":"2026-07-05T08:57:43.858922+00:00","updated_at":"2026-07-05T08:57:43.858922+00:00"}