{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MIK252VPMPLO3FMCGQGWO4JXB3","short_pith_number":"pith:MIK252VP","schema_version":"1.0","canonical_sha256":"6215aeeaaf63d6ed9582340d6771370ef433d93e97e67e9ca4efd663b4e74d36","source":{"kind":"arxiv","id":"2312.04461","version":1},"attestation_state":"computed","paper":{"title":"PhotoMaker: Customizing Realistic Human Photos via Stacked ID Embedding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Mingdeng Cao, Ming-Ming Cheng, Xintao Wang, Ying Shan, Zhen Li, Zhongang Qi","submitted_at":"2023-12-07T17:32:29Z","abstract_excerpt":"Recent advances in text-to-image generation have made remarkable progress in synthesizing realistic human photos conditioned on given text prompts. However, existing personalized generation methods cannot simultaneously satisfy the requirements of high efficiency, promising identity (ID) fidelity, and flexible text controllability. In this work, we introduce PhotoMaker, an efficient personalized text-to-image generation method, which mainly encodes an arbitrary number of input ID images into a stack ID embedding for preserving ID information. Such an embedding, serving as a unified ID represen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04461","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-07T17:32:29Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM"],"title_canon_sha256":"49a90278b16a470eb33c3e102859d8dc2dd6db873fbafc260ec96c646ab15b8f","abstract_canon_sha256":"b924aecb713c8f226b59a9c6cbf49f068a8737fc063fed5d004598f80de83da3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:34.794473Z","signature_b64":"/I40KcFDW2LuaXjVtidOq4fWXjydAkdblKfTAWaFM9HZsBVvx1VepH9QujA/a8JGH4/dL8+n1zJnpwzw9PXvBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6215aeeaaf63d6ed9582340d6771370ef433d93e97e67e9ca4efd663b4e74d36","last_reissued_at":"2026-07-05T07:21:34.794008Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:34.794008Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PhotoMaker: Customizing Realistic Human Photos via Stacked ID Embedding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Mingdeng Cao, Ming-Ming Cheng, Xintao Wang, Ying Shan, Zhen Li, Zhongang Qi","submitted_at":"2023-12-07T17:32:29Z","abstract_excerpt":"Recent advances in text-to-image generation have made remarkable progress in synthesizing realistic human photos conditioned on given text prompts. However, existing personalized generation methods cannot simultaneously satisfy the requirements of high efficiency, promising identity (ID) fidelity, and flexible text controllability. In this work, we introduce PhotoMaker, an efficient personalized text-to-image generation method, which mainly encodes an arbitrary number of input ID images into a stack ID embedding for preserving ID information. Such an embedding, serving as a unified ID represen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04461","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04461/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04461","created_at":"2026-07-05T07:21:34.794069+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04461v1","created_at":"2026-07-05T07:21:34.794069+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04461","created_at":"2026-07-05T07:21:34.794069+00:00"},{"alias_kind":"pith_short_12","alias_value":"MIK252VPMPLO","created_at":"2026-07-05T07:21:34.794069+00:00"},{"alias_kind":"pith_short_16","alias_value":"MIK252VPMPLO3FMC","created_at":"2026-07-05T07:21:34.794069+00:00"},{"alias_kind":"pith_short_8","alias_value":"MIK252VP","created_at":"2026-07-05T07:21:34.794069+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26930","citing_title":"PortraitGen: Exemplar-Driven GRPO with Dual-Reward Guidance for Photorealistic Portrait Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01383","citing_title":"MIBE: Multi-subject Interaction Benchmark and Evaluator for Personalized Image Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00607","citing_title":"IdGlow: Dynamic Identity Modulation for Multi-Subject Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08090","citing_title":"DSH-Bench: A Difficulty- and Scenario-Aware Benchmark with Hierarchical Subject Taxonomy for Subject-Driven Text-to-Image Generation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13720","citing_title":"Movie Gen: A Cast of Media Foundation Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07257","citing_title":"Adaptive Subspace Projection for Generative Personalization","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3","json":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3.json","graph_json":"https://pith.science/api/pith-number/MIK252VPMPLO3FMCGQGWO4JXB3/graph.json","events_json":"https://pith.science/api/pith-number/MIK252VPMPLO3FMCGQGWO4JXB3/events.json","paper":"https://pith.science/paper/MIK252VP"},"agent_actions":{"view_html":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3","download_json":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3.json","view_paper":"https://pith.science/paper/MIK252VP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04461&json=true","fetch_graph":"https://pith.science/api/pith-number/MIK252VPMPLO3FMCGQGWO4JXB3/graph.json","fetch_events":"https://pith.science/api/pith-number/MIK252VPMPLO3FMCGQGWO4JXB3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3/action/storage_attestation","attest_author":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3/action/author_attestation","sign_citation":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3/action/citation_signature","submit_replication":"https://pith.science/pith/MIK252VPMPLO3FMCGQGWO4JXB3/action/replication_record"}},"created_at":"2026-07-05T07:21:34.794069+00:00","updated_at":"2026-07-05T07:21:34.794069+00:00"}