{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JGYXKL6EGNYVV57IDR3OQHGUX6","short_pith_number":"pith:JGYXKL6E","schema_version":"1.0","canonical_sha256":"49b1752fc433715af7e81c76e81cd4bf84f9c8adfb91a8b4d9055d53004aead0","source":{"kind":"arxiv","id":"2410.20717","version":1},"attestation_state":"computed","paper":{"title":"Face-MLLM: A Large Face Perception Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haomiao Sun, Hu Han, Mingjie He, Shiguang Shan, Tianheng Lian","submitted_at":"2024-10-28T04:19:32Z","abstract_excerpt":"Although multimodal large language models (MLLMs) have achieved promising results on a wide range of vision-language tasks, their ability to perceive and understand human faces is rarely explored. In this work, we comprehensively evaluate existing MLLMs on face perception tasks. The quantitative results reveal that existing MLLMs struggle to handle these tasks. The primary reason is the lack of image-text datasets that contain fine-grained descriptions of human faces. To tackle this problem, we design a practical pipeline for constructing datasets, upon which we further build a novel multimoda"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.20717","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-28T04:19:32Z","cross_cats_sorted":[],"title_canon_sha256":"ee9896a6eb632211c6e6dc196d0481b562ad0951094526e2ce478f826481fd09","abstract_canon_sha256":"c90a7c4a7a4347c1a64e54c91f618771be680b385c9cfb79765a125d6e86b217"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:09.898023Z","signature_b64":"DhvfuRFvbn3QPFQNe6RnUc/lcMwrKJtbXHzIxndrCyN5oU0CaJDYZALgnX4WIEwV6eIFY4jDQ3E6Gwp6jAtaAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49b1752fc433715af7e81c76e81cd4bf84f9c8adfb91a8b4d9055d53004aead0","last_reissued_at":"2026-07-05T09:27:09.897565Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:09.897565Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Face-MLLM: A Large Face Perception Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haomiao Sun, Hu Han, Mingjie He, Shiguang Shan, Tianheng Lian","submitted_at":"2024-10-28T04:19:32Z","abstract_excerpt":"Although multimodal large language models (MLLMs) have achieved promising results on a wide range of vision-language tasks, their ability to perceive and understand human faces is rarely explored. In this work, we comprehensively evaluate existing MLLMs on face perception tasks. The quantitative results reveal that existing MLLMs struggle to handle these tasks. The primary reason is the lack of image-text datasets that contain fine-grained descriptions of human faces. To tackle this problem, we design a practical pipeline for constructing datasets, upon which we further build a novel multimoda"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.20717","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.20717/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.20717","created_at":"2026-07-05T09:27:09.897627+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.20717v1","created_at":"2026-07-05T09:27:09.897627+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.20717","created_at":"2026-07-05T09:27:09.897627+00:00"},{"alias_kind":"pith_short_12","alias_value":"JGYXKL6EGNYV","created_at":"2026-07-05T09:27:09.897627+00:00"},{"alias_kind":"pith_short_16","alias_value":"JGYXKL6EGNYVV57I","created_at":"2026-07-05T09:27:09.897627+00:00"},{"alias_kind":"pith_short_8","alias_value":"JGYXKL6E","created_at":"2026-07-05T09:27:09.897627+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.09158","citing_title":"FaVChat: Hierarchical Prompt-Query Guided Facial Video Understanding with Data-Efficient GRPO","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18073","citing_title":"FPBench: A Comprehensive Benchmark of Multimodal Large Language Models for Fingerprint Analysis","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6","json":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6.json","graph_json":"https://pith.science/api/pith-number/JGYXKL6EGNYVV57IDR3OQHGUX6/graph.json","events_json":"https://pith.science/api/pith-number/JGYXKL6EGNYVV57IDR3OQHGUX6/events.json","paper":"https://pith.science/paper/JGYXKL6E"},"agent_actions":{"view_html":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6","download_json":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6.json","view_paper":"https://pith.science/paper/JGYXKL6E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.20717&json=true","fetch_graph":"https://pith.science/api/pith-number/JGYXKL6EGNYVV57IDR3OQHGUX6/graph.json","fetch_events":"https://pith.science/api/pith-number/JGYXKL6EGNYVV57IDR3OQHGUX6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6/action/storage_attestation","attest_author":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6/action/author_attestation","sign_citation":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6/action/citation_signature","submit_replication":"https://pith.science/pith/JGYXKL6EGNYVV57IDR3OQHGUX6/action/replication_record"}},"created_at":"2026-07-05T09:27:09.897627+00:00","updated_at":"2026-07-05T09:27:09.897627+00:00"}