{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LXOVXU3VWAC7IVCK46J5LZJSFD","short_pith_number":"pith:LXOVXU3V","schema_version":"1.0","canonical_sha256":"5ddd5bd375b005f4544ae793d5e53228db2031484a912c7dc3c5dfb56acdf55e","source":{"kind":"arxiv","id":"2408.09698","version":5},"attestation_state":"computed","paper":{"title":"Harnessing Multimodal Large Language Models for Multimodal Sequential Recommendation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Hengruo Zhang, Hui Xiong, Kai Zhang, Peijun Zhu, Runlong Yu, Tianshu Wang, Yishan Shen, Yuyang Ye, Zhi Zheng","submitted_at":"2024-08-19T04:44:32Z","abstract_excerpt":"Recent advances in Large Language Models (LLMs) have demonstrated significant potential in the field of Recommendation Systems (RSs). Most existing studies have focused on converting user behavior logs into textual prompts and leveraging techniques such as prompt tuning to enable LLMs for recommendation tasks. Meanwhile, research interest has recently grown in multimodal recommendation systems that integrate data from images, text, and other sources using modality fusion techniques. This introduces new challenges to the existing LLM-based recommendation paradigm which relies solely on text mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.09698","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-08-19T04:44:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"01afef1ec85f60e8e2c557e90ba99129421fe60a9b5ccf9c123c4ea8e38ec12f","abstract_canon_sha256":"3695a1571d47e21ecd9c69d8512a817e7770dcd10a664f20f212049b2e4d9076"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:26.592702Z","signature_b64":"gY0H/6Fvg8S7DBu5QdnmMbhyMCCou1GvJvXdJEwSdDsyTYKTUBhbuDEfKS2SFTDal0YVrLGxQC36TxfAqBoCDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ddd5bd375b005f4544ae793d5e53228db2031484a912c7dc3c5dfb56acdf55e","last_reissued_at":"2026-07-05T10:00:26.592216Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:26.592216Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Harnessing Multimodal Large Language Models for Multimodal Sequential Recommendation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.IR","authors_text":"Hengruo Zhang, Hui Xiong, Kai Zhang, Peijun Zhu, Runlong Yu, Tianshu Wang, Yishan Shen, Yuyang Ye, Zhi Zheng","submitted_at":"2024-08-19T04:44:32Z","abstract_excerpt":"Recent advances in Large Language Models (LLMs) have demonstrated significant potential in the field of Recommendation Systems (RSs). Most existing studies have focused on converting user behavior logs into textual prompts and leveraging techniques such as prompt tuning to enable LLMs for recommendation tasks. Meanwhile, research interest has recently grown in multimodal recommendation systems that integrate data from images, text, and other sources using modality fusion techniques. This introduces new challenges to the existing LLM-based recommendation paradigm which relies solely on text mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.09698","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.09698/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.09698","created_at":"2026-07-05T10:00:26.592277+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.09698v5","created_at":"2026-07-05T10:00:26.592277+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.09698","created_at":"2026-07-05T10:00:26.592277+00:00"},{"alias_kind":"pith_short_12","alias_value":"LXOVXU3VWAC7","created_at":"2026-07-05T10:00:26.592277+00:00"},{"alias_kind":"pith_short_16","alias_value":"LXOVXU3VWAC7IVCK","created_at":"2026-07-05T10:00:26.592277+00:00"},{"alias_kind":"pith_short_8","alias_value":"LXOVXU3V","created_at":"2026-07-05T10:00:26.592277+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.21863","citing_title":"Frozen LVLMs for Micro-Video Recommendation: A Systematic Study of Feature Extraction and Fusion","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD","json":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD.json","graph_json":"https://pith.science/api/pith-number/LXOVXU3VWAC7IVCK46J5LZJSFD/graph.json","events_json":"https://pith.science/api/pith-number/LXOVXU3VWAC7IVCK46J5LZJSFD/events.json","paper":"https://pith.science/paper/LXOVXU3V"},"agent_actions":{"view_html":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD","download_json":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD.json","view_paper":"https://pith.science/paper/LXOVXU3V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.09698&json=true","fetch_graph":"https://pith.science/api/pith-number/LXOVXU3VWAC7IVCK46J5LZJSFD/graph.json","fetch_events":"https://pith.science/api/pith-number/LXOVXU3VWAC7IVCK46J5LZJSFD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD/action/storage_attestation","attest_author":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD/action/author_attestation","sign_citation":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD/action/citation_signature","submit_replication":"https://pith.science/pith/LXOVXU3VWAC7IVCK46J5LZJSFD/action/replication_record"}},"created_at":"2026-07-05T10:00:26.592277+00:00","updated_at":"2026-07-05T10:00:26.592277+00:00"}