{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:MIHXHUL4PQSERGMWUFY4O2ZQL4","short_pith_number":"pith:MIHXHUL4","schema_version":"1.0","canonical_sha256":"620f73d17c7c24489996a171c76b305f3b62ec55ab65c6dd13b3e404e3eeb748","source":{"kind":"arxiv","id":"2608.00215","version":1},"attestation_state":"computed","paper":{"title":"Personalizing Large Language Model Agents with Small Policy Models","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dian Jin, Doudou Zhou, Huichao Li, Rundong Huang, Yihe Pan, Zhi Zhang","submitted_at":"2026-07-31T18:56:30Z","abstract_excerpt":"Large language model (LLM) agents can retrieve memory, call tools, ask clarifying questions, and vary response style, yet adapting these execution decisions to an individual user remains difficult. Fine-tuning a separate LLM is costly or impossible for proprietary systems, while prompts and memory primarily expose user information to the agent rather than adapt its execution decisions from feedback. We formulate personalization of a frozen agent as online learning of a per-user execution policy from scalar feedback observed only for the executed action. We propose FABLE (Factorized Adaptive Ba"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.00215","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-31T18:56:30Z","cross_cats_sorted":[],"title_canon_sha256":"b88f42df4cbe70fc6b00404c57ee41e338573756edf26924c6b0e36b66ead17f","abstract_canon_sha256":"39c31555b122b29645e30fbaac791cef1577447234ddfbf5b1990c27dde50937"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-04T00:34:25.396469Z","signature_b64":"PpuMiV0vv8TeYeXuhDO+tYmtcrkufzN4WqsjYHp7P74WuU+0AssooJwbCUS4lREENWea1ieodrh3qH+BmEO5Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"620f73d17c7c24489996a171c76b305f3b62ec55ab65c6dd13b3e404e3eeb748","last_reissued_at":"2026-08-04T00:34:25.394896Z","signature_status":"signed_v1","first_computed_at":"2026-08-04T00:34:25.394896Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Personalizing Large Language Model Agents with Small Policy Models","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dian Jin, Doudou Zhou, Huichao Li, Rundong Huang, Yihe Pan, Zhi Zhang","submitted_at":"2026-07-31T18:56:30Z","abstract_excerpt":"Large language model (LLM) agents can retrieve memory, call tools, ask clarifying questions, and vary response style, yet adapting these execution decisions to an individual user remains difficult. Fine-tuning a separate LLM is costly or impossible for proprietary systems, while prompts and memory primarily expose user information to the agent rather than adapt its execution decisions from feedback. We formulate personalization of a frozen agent as online learning of a per-user execution policy from scalar feedback observed only for the executed action. We propose FABLE (Factorized Adaptive Ba"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.00215","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.00215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.00215","created_at":"2026-08-04T00:34:25.396143+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.00215v1","created_at":"2026-08-04T00:34:25.396143+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.00215","created_at":"2026-08-04T00:34:25.396143+00:00"},{"alias_kind":"pith_short_12","alias_value":"MIHXHUL4PQSE","created_at":"2026-08-04T00:34:25.396143+00:00"},{"alias_kind":"pith_short_16","alias_value":"MIHXHUL4PQSERGMW","created_at":"2026-08-04T00:34:25.396143+00:00"},{"alias_kind":"pith_short_8","alias_value":"MIHXHUL4","created_at":"2026-08-04T00:34:25.396143+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4","json":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4.json","graph_json":"https://pith.science/api/pith-number/MIHXHUL4PQSERGMWUFY4O2ZQL4/graph.json","events_json":"https://pith.science/api/pith-number/MIHXHUL4PQSERGMWUFY4O2ZQL4/events.json","paper":"https://pith.science/paper/MIHXHUL4"},"agent_actions":{"view_html":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4","download_json":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4.json","view_paper":"https://pith.science/paper/MIHXHUL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.00215&json=true","fetch_graph":"https://pith.science/api/pith-number/MIHXHUL4PQSERGMWUFY4O2ZQL4/graph.json","fetch_events":"https://pith.science/api/pith-number/MIHXHUL4PQSERGMWUFY4O2ZQL4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4/action/storage_attestation","attest_author":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4/action/author_attestation","sign_citation":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4/action/citation_signature","submit_replication":"https://pith.science/pith/MIHXHUL4PQSERGMWUFY4O2ZQL4/action/replication_record"}},"created_at":"2026-08-04T00:34:25.396143+00:00","updated_at":"2026-08-04T00:34:25.396143+00:00"}