{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TNF5HNI4SISZ6OWHS5TZVERMIX","short_pith_number":"pith:TNF5HNI4","schema_version":"1.0","canonical_sha256":"9b4bd3b51c92259f3ac797679a922c45f189dd2c0ab93fb2ecbe8e41043ad71b","source":{"kind":"arxiv","id":"2409.04421","version":2},"attestation_state":"computed","paper":{"title":"RLPF: Reinforcement Learning from Prediction Feedback for User Summarization with LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bradley Green, Chao Wang, Harrison Lee, Jiaxing Wu, Jun Xie, Lin Ning, Luyang Liu, Neo Wu, Shawn O'Banion, Sushant Prakash","submitted_at":"2024-09-06T17:30:45Z","abstract_excerpt":"LLM-powered personalization agent systems employ Large Language Models (LLMs) to predict users' behavior from their past activities. However, their effectiveness often hinges on the ability to effectively leverage extensive, long user historical data due to its inherent noise and length of such data. Existing pretrained LLMs may generate summaries that are concise but lack the necessary context for downstream tasks, hindering their utility in personalization systems. To address these challenges, we introduce Reinforcement Learning from Prediction Feedback (RLPF). RLPF fine-tunes LLMs to genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.04421","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-06T17:30:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2699beef68cae180a0d4026f72f6949789b72ebd6b81bfdbaa40e2863ddfb46a","abstract_canon_sha256":"2cccac77cf0d908805d8927240777f54d2da778ff240de9ce744e40772cceaf6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:06.100481Z","signature_b64":"KqHNx3zNrmZFnONCHQ2Tf641vKAMmZod7KJjNY+sVFFW3KcP1pbpIZfaXT9BkhneAJ3sjRPv1zqPm/fVoGC5DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b4bd3b51c92259f3ac797679a922c45f189dd2c0ab93fb2ecbe8e41043ad71b","last_reissued_at":"2026-07-05T10:02:06.100020Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:06.100020Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLPF: Reinforcement Learning from Prediction Feedback for User Summarization with LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bradley Green, Chao Wang, Harrison Lee, Jiaxing Wu, Jun Xie, Lin Ning, Luyang Liu, Neo Wu, Shawn O'Banion, Sushant Prakash","submitted_at":"2024-09-06T17:30:45Z","abstract_excerpt":"LLM-powered personalization agent systems employ Large Language Models (LLMs) to predict users' behavior from their past activities. However, their effectiveness often hinges on the ability to effectively leverage extensive, long user historical data due to its inherent noise and length of such data. Existing pretrained LLMs may generate summaries that are concise but lack the necessary context for downstream tasks, hindering their utility in personalization systems. To address these challenges, we introduce Reinforcement Learning from Prediction Feedback (RLPF). RLPF fine-tunes LLMs to genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04421","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.04421/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.04421","created_at":"2026-07-05T10:02:06.100079+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.04421v2","created_at":"2026-07-05T10:02:06.100079+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04421","created_at":"2026-07-05T10:02:06.100079+00:00"},{"alias_kind":"pith_short_12","alias_value":"TNF5HNI4SISZ","created_at":"2026-07-05T10:02:06.100079+00:00"},{"alias_kind":"pith_short_16","alias_value":"TNF5HNI4SISZ6OWH","created_at":"2026-07-05T10:02:06.100079+00:00"},{"alias_kind":"pith_short_8","alias_value":"TNF5HNI4","created_at":"2026-07-05T10:02:06.100079+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.15388","citing_title":"TrackRec: Iterative Alternating Feedback with Chain-of-Thought via Preference Alignment for Recommendation","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX","json":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX.json","graph_json":"https://pith.science/api/pith-number/TNF5HNI4SISZ6OWHS5TZVERMIX/graph.json","events_json":"https://pith.science/api/pith-number/TNF5HNI4SISZ6OWHS5TZVERMIX/events.json","paper":"https://pith.science/paper/TNF5HNI4"},"agent_actions":{"view_html":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX","download_json":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX.json","view_paper":"https://pith.science/paper/TNF5HNI4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.04421&json=true","fetch_graph":"https://pith.science/api/pith-number/TNF5HNI4SISZ6OWHS5TZVERMIX/graph.json","fetch_events":"https://pith.science/api/pith-number/TNF5HNI4SISZ6OWHS5TZVERMIX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX/action/storage_attestation","attest_author":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX/action/author_attestation","sign_citation":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX/action/citation_signature","submit_replication":"https://pith.science/pith/TNF5HNI4SISZ6OWHS5TZVERMIX/action/replication_record"}},"created_at":"2026-07-05T10:02:06.100079+00:00","updated_at":"2026-07-05T10:02:06.100079+00:00"}