{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U7IKCQNTLKDRZI6KTWIBLCK65W","short_pith_number":"pith:U7IKCQNT","schema_version":"1.0","canonical_sha256":"a7d0a141b35a871ca3ca9d9015895eedbfd71cce868fc59be91ad4ad0dbfaf34","source":{"kind":"arxiv","id":"2410.07113","version":1},"attestation_state":"computed","paper":{"title":"Personalized Visual Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jianshu Zhang, Jipeng Zhang, Renjie Pi, Rui Pan, Tianyang Han, Tong Zhang","submitted_at":"2024-10-09T17:46:53Z","abstract_excerpt":"Recent advancements in multimodal large language models (MLLMs) have demonstrated significant progress; however, these models exhibit a notable limitation, which we refer to as \"face blindness\". Specifically, they can engage in general conversations but fail to conduct personalized dialogues targeting at specific individuals. This deficiency hinders the application of MLLMs in personalized settings, such as tailored visual assistants on mobile devices, or domestic robots that need to recognize members of the family. In this paper, we introduce Personalized Visual Instruction Tuning (PVIT), a n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07113","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-09T17:46:53Z","cross_cats_sorted":[],"title_canon_sha256":"e9ae2a74cd15455da724db2594e02bc36d50d61fd78d0388c2fbe4e06666c403","abstract_canon_sha256":"029785cc1c62ca0e57aac120d02673374b5d62daea814318af10289fe29f2cbd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:18:11.042210Z","signature_b64":"0AE97H8DroyDo9MqmIHs0oazBC0lUlJzu6jdhR+xeWeMfckkGVfoWoa+J0wdCZalQ2VCScCNKyFFsJpWN/ofCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7d0a141b35a871ca3ca9d9015895eedbfd71cce868fc59be91ad4ad0dbfaf34","last_reissued_at":"2026-07-05T09:18:11.041760Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:18:11.041760Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Personalized Visual Instruction Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jianshu Zhang, Jipeng Zhang, Renjie Pi, Rui Pan, Tianyang Han, Tong Zhang","submitted_at":"2024-10-09T17:46:53Z","abstract_excerpt":"Recent advancements in multimodal large language models (MLLMs) have demonstrated significant progress; however, these models exhibit a notable limitation, which we refer to as \"face blindness\". Specifically, they can engage in general conversations but fail to conduct personalized dialogues targeting at specific individuals. This deficiency hinders the application of MLLMs in personalized settings, such as tailored visual assistants on mobile devices, or domestic robots that need to recognize members of the family. In this paper, we introduce Personalized Visual Instruction Tuning (PVIT), a n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07113","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07113/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07113","created_at":"2026-07-05T09:18:11.041815+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07113v1","created_at":"2026-07-05T09:18:11.041815+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07113","created_at":"2026-07-05T09:18:11.041815+00:00"},{"alias_kind":"pith_short_12","alias_value":"U7IKCQNTLKDR","created_at":"2026-07-05T09:18:11.041815+00:00"},{"alias_kind":"pith_short_16","alias_value":"U7IKCQNTLKDRZI6K","created_at":"2026-07-05T09:18:11.041815+00:00"},{"alias_kind":"pith_short_8","alias_value":"U7IKCQNT","created_at":"2026-07-05T09:18:11.041815+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00465","citing_title":"StochasT: Learning with Stochastic Turn Depth for Visual Instruction Tuning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31513","citing_title":"Personalize Your Large Vision-language Models With In-context Prompt Tuning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13074","citing_title":"PersonaVLM: Long-Term Personalized Multimodal LLMs","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10936","citing_title":"Personal Visual Context Learning in Large Multimodal Models","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W","json":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W.json","graph_json":"https://pith.science/api/pith-number/U7IKCQNTLKDRZI6KTWIBLCK65W/graph.json","events_json":"https://pith.science/api/pith-number/U7IKCQNTLKDRZI6KTWIBLCK65W/events.json","paper":"https://pith.science/paper/U7IKCQNT"},"agent_actions":{"view_html":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W","download_json":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W.json","view_paper":"https://pith.science/paper/U7IKCQNT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07113&json=true","fetch_graph":"https://pith.science/api/pith-number/U7IKCQNTLKDRZI6KTWIBLCK65W/graph.json","fetch_events":"https://pith.science/api/pith-number/U7IKCQNTLKDRZI6KTWIBLCK65W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W/action/storage_attestation","attest_author":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W/action/author_attestation","sign_citation":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W/action/citation_signature","submit_replication":"https://pith.science/pith/U7IKCQNTLKDRZI6KTWIBLCK65W/action/replication_record"}},"created_at":"2026-07-05T09:18:11.041815+00:00","updated_at":"2026-07-05T09:18:11.041815+00:00"}