{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JFRBLTJJORDPPNOZSXMYILW2WI","short_pith_number":"pith:JFRBLTJJ","schema_version":"1.0","canonical_sha256":"496215cd297446f7b5d995d9842edab20edbd77632c50a9b618e25876f01df54","source":{"kind":"arxiv","id":"2403.02302","version":4},"attestation_state":"computed","paper":{"title":"Beyond Specialization: Assessing the Capabilities of MLLMs in Age and Gender Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Grigorii Alekseenko, Irina Tolstykh, Maksim Kuprashevich","submitted_at":"2024-03-04T18:32:12Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have recently gained immense popularity. Powerful commercial models like ChatGPT-4V and Gemini, as well as open-source ones such as LLaVA, are essentially general-purpose models and are applied to solve a wide variety of tasks, including those in computer vision. These neural networks possess such strong general knowledge and reasoning abilities that they have proven capable of working even on tasks for which they were not specifically trained. We compared the capabilities of the most powerful MLLMs to date: ShareGPT4V, ChatGPT, LLaVA-Next in a speciali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.02302","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-04T18:32:12Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8b056cb8cf8f46cf477571cf1c981355472d6fcaa962ab4ac1d9632f66d3e09f","abstract_canon_sha256":"899546631e8bea46822f45856835128046aa44b00bd9f76e6a4efb39585bfeb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:03:24.195333Z","signature_b64":"yzO8GoO3yOU5o/NUZPDj3FDMeTdh/WDtQHGIU+1wO2oWs/e3c+6MVi6ZZgFeDzYAjB/+GPIRvscpJFK7B009Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"496215cd297446f7b5d995d9842edab20edbd77632c50a9b618e25876f01df54","last_reissued_at":"2026-07-05T10:03:24.194864Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:03:24.194864Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Specialization: Assessing the Capabilities of MLLMs in Age and Gender Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Grigorii Alekseenko, Irina Tolstykh, Maksim Kuprashevich","submitted_at":"2024-03-04T18:32:12Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have recently gained immense popularity. Powerful commercial models like ChatGPT-4V and Gemini, as well as open-source ones such as LLaVA, are essentially general-purpose models and are applied to solve a wide variety of tasks, including those in computer vision. These neural networks possess such strong general knowledge and reasoning abilities that they have proven capable of working even on tasks for which they were not specifically trained. We compared the capabilities of the most powerful MLLMs to date: ShareGPT4V, ChatGPT, LLaVA-Next in a speciali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.02302","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.02302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.02302","created_at":"2026-07-05T10:03:24.194921+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.02302v4","created_at":"2026-07-05T10:03:24.194921+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.02302","created_at":"2026-07-05T10:03:24.194921+00:00"},{"alias_kind":"pith_short_12","alias_value":"JFRBLTJJORDP","created_at":"2026-07-05T10:03:24.194921+00:00"},{"alias_kind":"pith_short_16","alias_value":"JFRBLTJJORDPPNOZ","created_at":"2026-07-05T10:03:24.194921+00:00"},{"alias_kind":"pith_short_8","alias_value":"JFRBLTJJ","created_at":"2026-07-05T10:03:24.194921+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.14799","citing_title":"Analyzing Character Representation in Media Content using Multimodal Foundation Model: Effectiveness and Trust","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI","json":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI.json","graph_json":"https://pith.science/api/pith-number/JFRBLTJJORDPPNOZSXMYILW2WI/graph.json","events_json":"https://pith.science/api/pith-number/JFRBLTJJORDPPNOZSXMYILW2WI/events.json","paper":"https://pith.science/paper/JFRBLTJJ"},"agent_actions":{"view_html":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI","download_json":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI.json","view_paper":"https://pith.science/paper/JFRBLTJJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.02302&json=true","fetch_graph":"https://pith.science/api/pith-number/JFRBLTJJORDPPNOZSXMYILW2WI/graph.json","fetch_events":"https://pith.science/api/pith-number/JFRBLTJJORDPPNOZSXMYILW2WI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI/action/storage_attestation","attest_author":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI/action/author_attestation","sign_citation":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI/action/citation_signature","submit_replication":"https://pith.science/pith/JFRBLTJJORDPPNOZSXMYILW2WI/action/replication_record"}},"created_at":"2026-07-05T10:03:24.194921+00:00","updated_at":"2026-07-05T10:03:24.194921+00:00"}