{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6EAATE725QFNFMG45YMSGHZLTL","short_pith_number":"pith:6EAATE72","schema_version":"1.0","canonical_sha256":"f1000993faec0ad2b0dcee19231f2b9ac52f616a3eca5babd056986940eaf068","source":{"kind":"arxiv","id":"2507.10300","version":1},"attestation_state":"computed","paper":{"title":"FaceLLM: A Multimodal Large Language Model for Face Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Hatef Otroshi Shahreza, S\\'ebastien Marcel","submitted_at":"2025-07-14T14:04:14Z","abstract_excerpt":"Multimodal large language models (MLLMs) have shown remarkable performance in vision-language tasks. However, existing MLLMs are primarily trained on generic datasets, limiting their ability to reason on domain-specific visual cues such as those in facial images. In particular, tasks that require detailed understanding of facial structure, expression, emotion, and demographic features remain underexplored by MLLMs due to the lack of large-scale annotated face image-text datasets. In this work, we introduce FaceLLM, a multimodal large language model trained specifically for facial image underst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.10300","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-14T14:04:14Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"108dc21d076ab051686b475a9a0b8b24fff47cfc2dd09defad4268506901a4c1","abstract_canon_sha256":"b7207e213cd02af8960f235015bdd45f7374c0b09c0a2be4924e223cf4447e6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:48.992662Z","signature_b64":"8+EQ33XklS95P6foqWk8rDpeGJa2Jn8HJV57rFH3eUsRDoWZGyA3IRX1CaA2eqGnesLJjqrUepS0sG4h+A/3Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1000993faec0ad2b0dcee19231f2b9ac52f616a3eca5babd056986940eaf068","last_reissued_at":"2026-07-05T11:36:48.992110Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:48.992110Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FaceLLM: A Multimodal Large Language Model for Face Understanding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Hatef Otroshi Shahreza, S\\'ebastien Marcel","submitted_at":"2025-07-14T14:04:14Z","abstract_excerpt":"Multimodal large language models (MLLMs) have shown remarkable performance in vision-language tasks. However, existing MLLMs are primarily trained on generic datasets, limiting their ability to reason on domain-specific visual cues such as those in facial images. In particular, tasks that require detailed understanding of facial structure, expression, emotion, and demographic features remain underexplored by MLLMs due to the lack of large-scale annotated face image-text datasets. In this work, we introduce FaceLLM, a multimodal large language model trained specifically for facial image underst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.10300","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.10300/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.10300","created_at":"2026-07-05T11:36:48.992171+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.10300v1","created_at":"2026-07-05T11:36:48.992171+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.10300","created_at":"2026-07-05T11:36:48.992171+00:00"},{"alias_kind":"pith_short_12","alias_value":"6EAATE725QFN","created_at":"2026-07-05T11:36:48.992171+00:00"},{"alias_kind":"pith_short_16","alias_value":"6EAATE725QFNFMG4","created_at":"2026-07-05T11:36:48.992171+00:00"},{"alias_kind":"pith_short_8","alias_value":"6EAATE72","created_at":"2026-07-05T11:36:48.992171+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.18073","citing_title":"FPBench: A Comprehensive Benchmark of Multimodal Large Language Models for Fingerprint Analysis","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL","json":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL.json","graph_json":"https://pith.science/api/pith-number/6EAATE725QFNFMG45YMSGHZLTL/graph.json","events_json":"https://pith.science/api/pith-number/6EAATE725QFNFMG45YMSGHZLTL/events.json","paper":"https://pith.science/paper/6EAATE72"},"agent_actions":{"view_html":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL","download_json":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL.json","view_paper":"https://pith.science/paper/6EAATE72","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.10300&json=true","fetch_graph":"https://pith.science/api/pith-number/6EAATE725QFNFMG45YMSGHZLTL/graph.json","fetch_events":"https://pith.science/api/pith-number/6EAATE725QFNFMG45YMSGHZLTL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL/action/storage_attestation","attest_author":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL/action/author_attestation","sign_citation":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL/action/citation_signature","submit_replication":"https://pith.science/pith/6EAATE725QFNFMG45YMSGHZLTL/action/replication_record"}},"created_at":"2026-07-05T11:36:48.992171+00:00","updated_at":"2026-07-05T11:36:48.992171+00:00"}