{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BQKL5GIH6INWZL5NRWNHFLU6QD","short_pith_number":"pith:BQKL5GIH","schema_version":"1.0","canonical_sha256":"0c14be9907f21b6cafad8d9a72ae9e80e55097a6de956ef2f1338a05f8d4ccb2","source":{"kind":"arxiv","id":"2503.19900","version":1},"attestation_state":"computed","paper":{"title":"CAFe: Unifying Representation and Generation with Contrastive-Autoregressive Finetuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Baosheng He, Hanchao Yu, Hao Yu, Jianyu Wang, Jiayi Liu, Lizhu Zhang, Lukasz Korycki, Shen Yan, Xiangjun Fan, Zhuokai Zhao","submitted_at":"2025-03-25T17:57:17Z","abstract_excerpt":"The rapid advancement of large vision-language models (LVLMs) has driven significant progress in multimodal tasks, enabling models to interpret, reason, and generate outputs across both visual and textual domains. While excelling in generative tasks, existing LVLMs often face limitations in tasks requiring high-fidelity representation learning, such as generating image or text embeddings for retrieval. Recent work has proposed finetuning LVLMs for representational learning, but the fine-tuned model often loses its generative capabilities due to the representational learning training paradigm. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.19900","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-25T17:57:17Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"320557c3f368f46022ca1806c2bbd310ea5d0bc4199666cec8e65d1166fe7f13","abstract_canon_sha256":"9b459f0edf4b3cb10d7893939b79243126e068cd7314d2753326254500d3a95e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:03.330293Z","signature_b64":"L1uqNG4mdeIyeWjhICI5r/y2LWZetlR+ifLYPKIzzqery8RFLmL6rz7+ojaBfV4m1PkerC/yb3crMExdDRnqDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c14be9907f21b6cafad8d9a72ae9e80e55097a6de956ef2f1338a05f8d4ccb2","last_reissued_at":"2026-07-05T10:39:03.329811Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:03.329811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CAFe: Unifying Representation and Generation with Contrastive-Autoregressive Finetuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Baosheng He, Hanchao Yu, Hao Yu, Jianyu Wang, Jiayi Liu, Lizhu Zhang, Lukasz Korycki, Shen Yan, Xiangjun Fan, Zhuokai Zhao","submitted_at":"2025-03-25T17:57:17Z","abstract_excerpt":"The rapid advancement of large vision-language models (LVLMs) has driven significant progress in multimodal tasks, enabling models to interpret, reason, and generate outputs across both visual and textual domains. While excelling in generative tasks, existing LVLMs often face limitations in tasks requiring high-fidelity representation learning, such as generating image or text embeddings for retrieval. Recent work has proposed finetuning LVLMs for representational learning, but the fine-tuned model often loses its generative capabilities due to the representational learning training paradigm. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.19900","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.19900/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.19900","created_at":"2026-07-05T10:39:03.329869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.19900v1","created_at":"2026-07-05T10:39:03.329869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.19900","created_at":"2026-07-05T10:39:03.329869+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQKL5GIH6INW","created_at":"2026-07-05T10:39:03.329869+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQKL5GIH6INWZL5N","created_at":"2026-07-05T10:39:03.329869+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQKL5GIH","created_at":"2026-07-05T10:39:03.329869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24621","citing_title":"FreeRet: MLLMs as Training-Free Retrievers","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21262","citing_title":"CausalEmbed: Auto-Regressive Multi-Vector Generation in Latent Space for Visual Document Embedding","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2602.19549","citing_title":"Sculpting the Vector Space: Towards Efficient Multi-Vector Visual Document Retrieval via Prune-then-Merge Framework","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02073","citing_title":"PLUME: Latent Reasoning Based Universal Multimodal Embedding","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22280","citing_title":"Beyond Chain-of-Thought: Rewrite as a Universal Interface for Generative Multimodal Embeddings","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11095","citing_title":"Bottleneck Tokens for Unified Multimodal Retrieval","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD","json":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD.json","graph_json":"https://pith.science/api/pith-number/BQKL5GIH6INWZL5NRWNHFLU6QD/graph.json","events_json":"https://pith.science/api/pith-number/BQKL5GIH6INWZL5NRWNHFLU6QD/events.json","paper":"https://pith.science/paper/BQKL5GIH"},"agent_actions":{"view_html":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD","download_json":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD.json","view_paper":"https://pith.science/paper/BQKL5GIH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.19900&json=true","fetch_graph":"https://pith.science/api/pith-number/BQKL5GIH6INWZL5NRWNHFLU6QD/graph.json","fetch_events":"https://pith.science/api/pith-number/BQKL5GIH6INWZL5NRWNHFLU6QD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD/action/storage_attestation","attest_author":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD/action/author_attestation","sign_citation":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD/action/citation_signature","submit_replication":"https://pith.science/pith/BQKL5GIH6INWZL5NRWNHFLU6QD/action/replication_record"}},"created_at":"2026-07-05T10:39:03.329869+00:00","updated_at":"2026-07-05T10:39:03.329869+00:00"}