{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O2NKKWXOT5K5F3FFSPPA5RTP3F","short_pith_number":"pith:O2NKKWXO","schema_version":"1.0","canonical_sha256":"769aa55aee9f55d2eca593de0ec66fd9423cbc60239a72019203d4e2c80dc6a8","source":{"kind":"arxiv","id":"2502.00633","version":1},"attestation_state":"computed","paper":{"title":"Lipschitz Lifelong Monte Carlo Tree Search for Mastering Non-Stationary Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Tian Lan, Zuyuan Zhang","submitted_at":"2025-02-02T02:45:20Z","abstract_excerpt":"Monte Carlo Tree Search (MCTS) has proven highly effective in solving complex planning tasks by balancing exploration and exploitation using Upper Confidence Bound for Trees (UCT). However, existing work have not considered MCTS-based lifelong planning, where an agent faces a non-stationary series of tasks -- e.g., with varying transition probabilities and rewards -- that are drawn sequentially throughout the operational lifetime. This paper presents LiZero for Lipschitz lifelong planning using MCTS. We propose a novel concept of adaptive UCT (aUCT) to transfer knowledge from a source task to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.00633","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-02T02:45:20Z","cross_cats_sorted":[],"title_canon_sha256":"b695a4c80f8e47201cddb0d960ba4a79c0d800ac34a8220233885fe3c9f1aa84","abstract_canon_sha256":"1979d7aa5ea0ef73e066cb9e387cb4ecd402bba49ea72a5597c18475e1aa2f30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:36.444466Z","signature_b64":"tAXv20cULKJiIe0tATCKBmYFv05Xb8lhu7qbLJjV0XALTS8piFGlex58fy0FOEJ3DMyNeHTVwfr07T6eABgACw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"769aa55aee9f55d2eca593de0ec66fd9423cbc60239a72019203d4e2c80dc6a8","last_reissued_at":"2026-07-05T10:08:36.444030Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:36.444030Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lipschitz Lifelong Monte Carlo Tree Search for Mastering Non-Stationary Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Tian Lan, Zuyuan Zhang","submitted_at":"2025-02-02T02:45:20Z","abstract_excerpt":"Monte Carlo Tree Search (MCTS) has proven highly effective in solving complex planning tasks by balancing exploration and exploitation using Upper Confidence Bound for Trees (UCT). However, existing work have not considered MCTS-based lifelong planning, where an agent faces a non-stationary series of tasks -- e.g., with varying transition probabilities and rewards -- that are drawn sequentially throughout the operational lifetime. This paper presents LiZero for Lipschitz lifelong planning using MCTS. We propose a novel concept of adaptive UCT (aUCT) to transfer knowledge from a source task to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.00633","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.00633/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.00633","created_at":"2026-07-05T10:08:36.444086+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.00633v1","created_at":"2026-07-05T10:08:36.444086+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.00633","created_at":"2026-07-05T10:08:36.444086+00:00"},{"alias_kind":"pith_short_12","alias_value":"O2NKKWXOT5K5","created_at":"2026-07-05T10:08:36.444086+00:00"},{"alias_kind":"pith_short_16","alias_value":"O2NKKWXOT5K5F3FF","created_at":"2026-07-05T10:08:36.444086+00:00"},{"alias_kind":"pith_short_8","alias_value":"O2NKKWXO","created_at":"2026-07-05T10:08:36.444086+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18809","citing_title":"Metric-Gradient Projection for Stable Multi-Agent Policy Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05048","citing_title":"MINT: Minimal Information Neuro-Symbolic Tree for Objective-Driven Knowledge-Gap Reasoning and Active Elicitation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06500","citing_title":"Operator-Guided Invariance Learning for Continuous Reinforcement Learning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F","json":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F.json","graph_json":"https://pith.science/api/pith-number/O2NKKWXOT5K5F3FFSPPA5RTP3F/graph.json","events_json":"https://pith.science/api/pith-number/O2NKKWXOT5K5F3FFSPPA5RTP3F/events.json","paper":"https://pith.science/paper/O2NKKWXO"},"agent_actions":{"view_html":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F","download_json":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F.json","view_paper":"https://pith.science/paper/O2NKKWXO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.00633&json=true","fetch_graph":"https://pith.science/api/pith-number/O2NKKWXOT5K5F3FFSPPA5RTP3F/graph.json","fetch_events":"https://pith.science/api/pith-number/O2NKKWXOT5K5F3FFSPPA5RTP3F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F/action/storage_attestation","attest_author":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F/action/author_attestation","sign_citation":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F/action/citation_signature","submit_replication":"https://pith.science/pith/O2NKKWXOT5K5F3FFSPPA5RTP3F/action/replication_record"}},"created_at":"2026-07-05T10:08:36.444086+00:00","updated_at":"2026-07-05T10:08:36.444086+00:00"}