{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XM7JYNT26LQUVDPFXB2LRUQH44","short_pith_number":"pith:XM7JYNT2","schema_version":"1.0","canonical_sha256":"bb3e9c367af2e14a8de5b874b8d207e70886bd11a48745dd0cb29b0aa5a551e4","source":{"kind":"arxiv","id":"2402.08670","version":1},"attestation_state":"computed","paper":{"title":"Rec-GPT4V: Multimodal Recommendation with Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lichao Sun, Philip S. Yu, Yuqing Liu, Yu Wang","submitted_at":"2024-02-13T18:51:18Z","abstract_excerpt":"The development of large vision-language models (LVLMs) offers the potential to address challenges faced by traditional multimodal recommendations thanks to their proficient understanding of static images and textual dynamics. However, the application of LVLMs in this field is still limited due to the following complexities: First, LVLMs lack user preference knowledge as they are trained from vast general datasets. Second, LVLMs suffer setbacks in addressing multiple image dynamics in scenarios involving discrete, noisy, and redundant image sequences. To overcome these issues, we propose the n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08670","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-13T18:51:18Z","cross_cats_sorted":[],"title_canon_sha256":"bd841b70758ee0cfd5e253df0bc026093dc3868765338a343f6eb641a2f2a084","abstract_canon_sha256":"604f68e242d26bb4f3b1ab16e63c33341625b8f9e2dc17e4f4eb77e386830302"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:44:49.410880Z","signature_b64":"iBvOxlWJ+z8fWWJ8RYgOJF8+aWLMwWBF0EGg76YsMLoX9QWrXeG+y4dEpWx07V4CKXOHCxjlMsUWtAVCbVBQCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb3e9c367af2e14a8de5b874b8d207e70886bd11a48745dd0cb29b0aa5a551e4","last_reissued_at":"2026-07-05T07:44:49.410376Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:44:49.410376Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rec-GPT4V: Multimodal Recommendation with Large Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lichao Sun, Philip S. Yu, Yuqing Liu, Yu Wang","submitted_at":"2024-02-13T18:51:18Z","abstract_excerpt":"The development of large vision-language models (LVLMs) offers the potential to address challenges faced by traditional multimodal recommendations thanks to their proficient understanding of static images and textual dynamics. However, the application of LVLMs in this field is still limited due to the following complexities: First, LVLMs lack user preference knowledge as they are trained from vast general datasets. Second, LVLMs suffer setbacks in addressing multiple image dynamics in scenarios involving discrete, noisy, and redundant image sequences. To overcome these issues, we propose the n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08670","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08670/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08670","created_at":"2026-07-05T07:44:49.410453+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08670v1","created_at":"2026-07-05T07:44:49.410453+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08670","created_at":"2026-07-05T07:44:49.410453+00:00"},{"alias_kind":"pith_short_12","alias_value":"XM7JYNT26LQU","created_at":"2026-07-05T07:44:49.410453+00:00"},{"alias_kind":"pith_short_16","alias_value":"XM7JYNT26LQUVDPF","created_at":"2026-07-05T07:44:49.410453+00:00"},{"alias_kind":"pith_short_8","alias_value":"XM7JYNT2","created_at":"2026-07-05T07:44:49.410453+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09595","citing_title":"Popcorn: A Configurable Benchmark for Visual Evidence in Multimodal Movie Recommendation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26941","citing_title":"The 2nd EReL@MIR Workshop on Efficient Representation Learning for Multimodal Information Retrieval","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15203","citing_title":"Agent4POI: Agentic Context-Conditioned Affordance Reasoning for Multimodal Point-of-Interest Recommendation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.27157","citing_title":"A Survey on Generative Recommendation: Data, Model, and Tasks","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18740","citing_title":"Multimodal Large Language Models with Adaptive Preference Optimization for Sequential Recommendation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21863","citing_title":"Frozen LVLMs for Micro-Video Recommendation: A Systematic Study of Feature Extraction and Fusion","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26247","citing_title":"TimeMM: Time-as-Operator Spectral Filtering for Dynamic Multimodal Recommendation","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44","json":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44.json","graph_json":"https://pith.science/api/pith-number/XM7JYNT26LQUVDPFXB2LRUQH44/graph.json","events_json":"https://pith.science/api/pith-number/XM7JYNT26LQUVDPFXB2LRUQH44/events.json","paper":"https://pith.science/paper/XM7JYNT2"},"agent_actions":{"view_html":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44","download_json":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44.json","view_paper":"https://pith.science/paper/XM7JYNT2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08670&json=true","fetch_graph":"https://pith.science/api/pith-number/XM7JYNT26LQUVDPFXB2LRUQH44/graph.json","fetch_events":"https://pith.science/api/pith-number/XM7JYNT26LQUVDPFXB2LRUQH44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44/action/storage_attestation","attest_author":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44/action/author_attestation","sign_citation":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44/action/citation_signature","submit_replication":"https://pith.science/pith/XM7JYNT26LQUVDPFXB2LRUQH44/action/replication_record"}},"created_at":"2026-07-05T07:44:49.410453+00:00","updated_at":"2026-07-05T07:44:49.410453+00:00"}