{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BYV5N55KJVTT5PLUZ4CNTU3VCH","short_pith_number":"pith:BYV5N55K","schema_version":"1.0","canonical_sha256":"0e2bd6f7aa4d673ebd74cf04d9d37511ee86dd927e777e3e8cd49b4028395e0f","source":{"kind":"arxiv","id":"2410.05165","version":3},"attestation_state":"computed","paper":{"title":"Efficient Inference for Large Language Model-based Generative Recommendation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Chaoqun Yang, Cunxiao Du, Fuli Feng, See-kiong Ng, Tat-Seng Chua, Wenjie Wang, Xinyu Lin, Yongqi Li","submitted_at":"2024-10-07T16:23:36Z","abstract_excerpt":"Large Language Model (LLM)-based generative recommendation has achieved notable success, yet its practical deployment is costly particularly due to excessive inference latency caused by autoregressive decoding. For lossless LLM decoding acceleration, Speculative Decoding (SD) has emerged as a promising solution. However, applying SD to generative recommendation presents unique challenges due to the requirement of generating top-K items (i.e., K distinct token sequences) as a recommendation list by beam search. This leads to more stringent verification in SD, where all the top-K sequences from "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05165","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-10-07T16:23:36Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d426a777ae288d4bd76a229888583448fb745688cc4b9e496a6fcf7b5f1d8c48","abstract_canon_sha256":"45b7a39e5c16934d9856c0c3f60d01d575f1dd512c47b8568a5a0e79ae6b2b03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:16.010326Z","signature_b64":"mYTKC2YY7Mp52pmwlgoG0oqBRvKHmiwHRU14PCLU9MU03f1gNhnbslEdrZI7q6f/erU3RXT2kD4RNpJlSOhaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e2bd6f7aa4d673ebd74cf04d9d37511ee86dd927e777e3e8cd49b4028395e0f","last_reissued_at":"2026-07-05T10:20:16.009794Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:16.009794Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Inference for Large Language Model-based Generative Recommendation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Chaoqun Yang, Cunxiao Du, Fuli Feng, See-kiong Ng, Tat-Seng Chua, Wenjie Wang, Xinyu Lin, Yongqi Li","submitted_at":"2024-10-07T16:23:36Z","abstract_excerpt":"Large Language Model (LLM)-based generative recommendation has achieved notable success, yet its practical deployment is costly particularly due to excessive inference latency caused by autoregressive decoding. For lossless LLM decoding acceleration, Speculative Decoding (SD) has emerged as a promising solution. However, applying SD to generative recommendation presents unique challenges due to the requirement of generating top-K items (i.e., K distinct token sequences) as a recommendation list by beam search. This leads to more stringent verification in SD, where all the top-K sequences from "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05165","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05165/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05165","created_at":"2026-07-05T10:20:16.009855+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05165v3","created_at":"2026-07-05T10:20:16.009855+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05165","created_at":"2026-07-05T10:20:16.009855+00:00"},{"alias_kind":"pith_short_12","alias_value":"BYV5N55KJVTT","created_at":"2026-07-05T10:20:16.009855+00:00"},{"alias_kind":"pith_short_16","alias_value":"BYV5N55KJVTT5PLU","created_at":"2026-07-05T10:20:16.009855+00:00"},{"alias_kind":"pith_short_8","alias_value":"BYV5N55K","created_at":"2026-07-05T10:20:16.009855+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11447","citing_title":"Conditional Memory Enhanced Item Representation for Generative Recommendation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05329","citing_title":"Semantic Trimming and Auxiliary Multi-step Prediction for Generative Recommendation","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH","json":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH.json","graph_json":"https://pith.science/api/pith-number/BYV5N55KJVTT5PLUZ4CNTU3VCH/graph.json","events_json":"https://pith.science/api/pith-number/BYV5N55KJVTT5PLUZ4CNTU3VCH/events.json","paper":"https://pith.science/paper/BYV5N55K"},"agent_actions":{"view_html":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH","download_json":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH.json","view_paper":"https://pith.science/paper/BYV5N55K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05165&json=true","fetch_graph":"https://pith.science/api/pith-number/BYV5N55KJVTT5PLUZ4CNTU3VCH/graph.json","fetch_events":"https://pith.science/api/pith-number/BYV5N55KJVTT5PLUZ4CNTU3VCH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH/action/storage_attestation","attest_author":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH/action/author_attestation","sign_citation":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH/action/citation_signature","submit_replication":"https://pith.science/pith/BYV5N55KJVTT5PLUZ4CNTU3VCH/action/replication_record"}},"created_at":"2026-07-05T10:20:16.009855+00:00","updated_at":"2026-07-05T10:20:16.009855+00:00"}