{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PRDPBD7OTG53UUEVGTU2JZV5E5","short_pith_number":"pith:PRDPBD7O","schema_version":"1.0","canonical_sha256":"7c46f08fee99bbba509534e9a4e6bd27531bffb40e9636c9cb566567671f6690","source":{"kind":"arxiv","id":"2401.13201","version":3},"attestation_state":"computed","paper":{"title":"MLLMReID: Multimodal Large Language Model-based Person Re-identification","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Shan Yang, Yongfei Zhang","submitted_at":"2024-01-24T03:07:26Z","abstract_excerpt":"Multimodal large language models (MLLM) have achieved satisfactory results in many tasks. However, their performance in the task of ReID (ReID) has not been explored to date. This paper will investigate how to adapt them for the task of ReID. An intuitive idea is to fine-tune MLLM with ReID image-text datasets, and then use their visual encoder as a backbone for ReID. However, there still exist two apparent issues: (1) Designing instructions for ReID, MLLMs may overfit specific instructions, and designing a variety of instructions will lead to higher costs. (2) When fine-tuning the visual enco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.13201","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-24T03:07:26Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7bb0ce74673c70596a1daac0cd25147148cd17bba415eee6ac26f16beb40ba2b","abstract_canon_sha256":"11f6af3664451cd907b568764c63eddc8320aebf04bc8d61eeeaeeacc66863f9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:20.568416Z","signature_b64":"8fX3zzzMltLgTAKWl6OFa8qG7Ba0YiQZe12ntvZkYtMPUcu1KU994k7HdcDK1adLFEGCYhIqmgROt9BqrxL7Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c46f08fee99bbba509534e9a4e6bd27531bffb40e9636c9cb566567671f6690","last_reissued_at":"2026-07-05T08:29:20.567924Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:20.567924Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLLMReID: Multimodal Large Language Model-based Person Re-identification","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Shan Yang, Yongfei Zhang","submitted_at":"2024-01-24T03:07:26Z","abstract_excerpt":"Multimodal large language models (MLLM) have achieved satisfactory results in many tasks. However, their performance in the task of ReID (ReID) has not been explored to date. This paper will investigate how to adapt them for the task of ReID. An intuitive idea is to fine-tune MLLM with ReID image-text datasets, and then use their visual encoder as a backbone for ReID. However, there still exist two apparent issues: (1) Designing instructions for ReID, MLLMs may overfit specific instructions, and designing a variety of instructions will lead to higher costs. (2) When fine-tuning the visual enco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.13201","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.13201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.13201","created_at":"2026-07-05T08:29:20.567985+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.13201v3","created_at":"2026-07-05T08:29:20.567985+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.13201","created_at":"2026-07-05T08:29:20.567985+00:00"},{"alias_kind":"pith_short_12","alias_value":"PRDPBD7OTG53","created_at":"2026-07-05T08:29:20.567985+00:00"},{"alias_kind":"pith_short_16","alias_value":"PRDPBD7OTG53UUEV","created_at":"2026-07-05T08:29:20.567985+00:00"},{"alias_kind":"pith_short_8","alias_value":"PRDPBD7O","created_at":"2026-07-05T08:29:20.567985+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02724","citing_title":"AVTrack: Audio-Visual Tracking in Human-centric Complex Scenes","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18376","citing_title":"Towards Robust Text-to-Image Person Retrieval: Multi-View Reformulation for Semantic Compensation","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5","json":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5.json","graph_json":"https://pith.science/api/pith-number/PRDPBD7OTG53UUEVGTU2JZV5E5/graph.json","events_json":"https://pith.science/api/pith-number/PRDPBD7OTG53UUEVGTU2JZV5E5/events.json","paper":"https://pith.science/paper/PRDPBD7O"},"agent_actions":{"view_html":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5","download_json":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5.json","view_paper":"https://pith.science/paper/PRDPBD7O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.13201&json=true","fetch_graph":"https://pith.science/api/pith-number/PRDPBD7OTG53UUEVGTU2JZV5E5/graph.json","fetch_events":"https://pith.science/api/pith-number/PRDPBD7OTG53UUEVGTU2JZV5E5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5/action/storage_attestation","attest_author":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5/action/author_attestation","sign_citation":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5/action/citation_signature","submit_replication":"https://pith.science/pith/PRDPBD7OTG53UUEVGTU2JZV5E5/action/replication_record"}},"created_at":"2026-07-05T08:29:20.567985+00:00","updated_at":"2026-07-05T08:29:20.567985+00:00"}