{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:56JXPVMM6PTFXZWYAI6WEZBR5H","short_pith_number":"pith:56JXPVMM","schema_version":"1.0","canonical_sha256":"ef9377d58cf3e65be6d8023d626431e9f6e5c1785bedd1b10bec4d67652ce6bd","source":{"kind":"arxiv","id":"2502.09597","version":1},"attestation_state":"computed","paper":{"title":"Do LLMs Recognize Your Preferences? Evaluating Personalized Preference Following in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Devamanyu Hazarika, Kaixiang Lin, Mingyi Hong, Siyan Zhao, Yang Liu","submitted_at":"2025-02-13T18:52:03Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used as chatbots, yet their ability to personalize responses to user preferences remains limited. We introduce PrefEval, a benchmark for evaluating LLMs' ability to infer, memorize and adhere to user preferences in a long-context conversational setting. PrefEval comprises 3,000 manually curated user preference and query pairs spanning 20 topics. PrefEval contains user personalization or preference information in both explicit and implicit forms, and evaluates LLM performance using a generation and a classification task. With PrefEval, we evaluated "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.09597","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-13T18:52:03Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d719dc764a10e639af76d32c1deb22e83329389c36b253fa569bcdb59b9452d2","abstract_canon_sha256":"8b59278d1c907a1875ec3dc7001e6f381c47ef6df6b681651f6d81f298838cfc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:59.854092Z","signature_b64":"wSNXWbejNQ1XRWc8rzCVcIPmc6iIKYFRkh+xZ0vIRA+RZpQ97R72o6nt3xtC6U5aJ6Zoon8+BS0awKMyJzpMAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef9377d58cf3e65be6d8023d626431e9f6e5c1785bedd1b10bec4d67652ce6bd","last_reissued_at":"2026-07-05T10:13:59.853637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:59.853637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do LLMs Recognize Your Preferences? Evaluating Personalized Preference Following in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Devamanyu Hazarika, Kaixiang Lin, Mingyi Hong, Siyan Zhao, Yang Liu","submitted_at":"2025-02-13T18:52:03Z","abstract_excerpt":"Large Language Models (LLMs) are increasingly used as chatbots, yet their ability to personalize responses to user preferences remains limited. We introduce PrefEval, a benchmark for evaluating LLMs' ability to infer, memorize and adhere to user preferences in a long-context conversational setting. PrefEval comprises 3,000 manually curated user preference and query pairs spanning 20 topics. PrefEval contains user personalization or preference information in both explicit and implicit forms, and evaluates LLM performance using a generation and a classification task. With PrefEval, we evaluated "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.09597","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.09597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.09597","created_at":"2026-07-05T10:13:59.853694+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.09597v1","created_at":"2026-07-05T10:13:59.853694+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.09597","created_at":"2026-07-05T10:13:59.853694+00:00"},{"alias_kind":"pith_short_12","alias_value":"56JXPVMM6PTF","created_at":"2026-07-05T10:13:59.853694+00:00"},{"alias_kind":"pith_short_16","alias_value":"56JXPVMM6PTFXZWY","created_at":"2026-07-05T10:13:59.853694+00:00"},{"alias_kind":"pith_short_8","alias_value":"56JXPVMM","created_at":"2026-07-05T10:13:59.853694+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17453","citing_title":"MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13192","citing_title":"Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06055","citing_title":"When Should Memory Stay Silent: Measuring Memory-Use Boundaries in Memory-Augmented Conversational Agents","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2603.23231","citing_title":"PERMA: Benchmarking Personalized Memory Agents via Event-Driven Preference and Realistic Task Environments","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06204","citing_title":"SensorPersona: An LLM-Empowered System for Continual Persona Extraction from Longitudinal Mobile Sensor Streams","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2507.03724","citing_title":"MemOS: A Memory OS for AI System","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13074","citing_title":"PersonaVLM: Long-Term Personalized Multimodal LLMs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00702","citing_title":"Learning How and What to Memorize: Cognition-Inspired Two-Stage Optimization for Evolving Memory","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06134","citing_title":"MAESTRO: Adapting GUIs and Guiding Navigation with User Preferences in Conversational Agents with GUIs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17283","citing_title":"HorizonBench: Long-Horizon Personalization with Evolving Preferences","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H","json":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H.json","graph_json":"https://pith.science/api/pith-number/56JXPVMM6PTFXZWYAI6WEZBR5H/graph.json","events_json":"https://pith.science/api/pith-number/56JXPVMM6PTFXZWYAI6WEZBR5H/events.json","paper":"https://pith.science/paper/56JXPVMM"},"agent_actions":{"view_html":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H","download_json":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H.json","view_paper":"https://pith.science/paper/56JXPVMM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.09597&json=true","fetch_graph":"https://pith.science/api/pith-number/56JXPVMM6PTFXZWYAI6WEZBR5H/graph.json","fetch_events":"https://pith.science/api/pith-number/56JXPVMM6PTFXZWYAI6WEZBR5H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H/action/storage_attestation","attest_author":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H/action/author_attestation","sign_citation":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H/action/citation_signature","submit_replication":"https://pith.science/pith/56JXPVMM6PTFXZWYAI6WEZBR5H/action/replication_record"}},"created_at":"2026-07-05T10:13:59.853694+00:00","updated_at":"2026-07-05T10:13:59.853694+00:00"}