{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:L7Y5VORU66BTY2CDJH5HSMVKE5","short_pith_number":"pith:L7Y5VORU","schema_version":"1.0","canonical_sha256":"5ff1daba34f7833c684349fa7932aa276706707f538a27df5a46d710479e285e","source":{"kind":"arxiv","id":"2505.11942","version":3},"attestation_state":"computed","paper":{"title":"LifelongAgentBench: Evaluating LLM Agents as Lifelong Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Duzhen Zhang, Junhao Zheng, Le Song, Qianli Ma, Qiuke Li, Xidi Cai, Yingying Zhang, Zhongzhi Li","submitted_at":"2025-05-17T10:09:11Z","abstract_excerpt":"Lifelong learning is essential for intelligent agents operating in dynamic environments. Current large language model (LLM)-based agents, however, remain stateless and unable to accumulate or transfer knowledge over time. Existing benchmarks treat agents as static systems and fail to evaluate lifelong learning capabilities. We present LifelongAgentBench, the first unified benchmark designed to systematically assess the lifelong learning ability of LLM agents. It provides skill-grounded, interdependent tasks across three interactive environments, Database, Operating System, and Knowledge Graph,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11942","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-17T10:09:11Z","cross_cats_sorted":[],"title_canon_sha256":"d426f94d21b316f2cd6fa4d72370c34f76b600c19c187f023b3d54237312ec72","abstract_canon_sha256":"e9954b8d77e188d3f3e0c85c183758719b7276e00f6d05d4c06f6872c9accf47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:31.906697Z","signature_b64":"09JIis5cqcA73Z2M161Bn2FORzx+EJYhyjMygVntVhq0XNTyUXnaUNcb7V+jUxYkIQQthE+cKIilAEE9fumaBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ff1daba34f7833c684349fa7932aa276706707f538a27df5a46d710479e285e","last_reissued_at":"2026-07-05T11:12:31.906150Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:31.906150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LifelongAgentBench: Evaluating LLM Agents as Lifelong Learners","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Duzhen Zhang, Junhao Zheng, Le Song, Qianli Ma, Qiuke Li, Xidi Cai, Yingying Zhang, Zhongzhi Li","submitted_at":"2025-05-17T10:09:11Z","abstract_excerpt":"Lifelong learning is essential for intelligent agents operating in dynamic environments. Current large language model (LLM)-based agents, however, remain stateless and unable to accumulate or transfer knowledge over time. Existing benchmarks treat agents as static systems and fail to evaluate lifelong learning capabilities. We present LifelongAgentBench, the first unified benchmark designed to systematically assess the lifelong learning ability of LLM agents. It provides skill-grounded, interdependent tasks across three interactive environments, Database, Operating System, and Knowledge Graph,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11942","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11942/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11942","created_at":"2026-07-05T11:12:31.906223+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11942v3","created_at":"2026-07-05T11:12:31.906223+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11942","created_at":"2026-07-05T11:12:31.906223+00:00"},{"alias_kind":"pith_short_12","alias_value":"L7Y5VORU66BT","created_at":"2026-07-05T11:12:31.906223+00:00"},{"alias_kind":"pith_short_16","alias_value":"L7Y5VORU66BTY2CD","created_at":"2026-07-05T11:12:31.906223+00:00"},{"alias_kind":"pith_short_8","alias_value":"L7Y5VORU","created_at":"2026-07-05T11:12:31.906223+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24775","citing_title":"Are We Ready For An Agent-Native Memory System?","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18829","citing_title":"GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08656","citing_title":"From Player to Master: Enhancing Test-Time Learning of LLM Agents via Reinforcement Learning over Memory","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08348","citing_title":"Bayesian-Agent: Posterior-Guided Skill Evolution for LLM Agent Harnesses","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06960","citing_title":"Tree-of-Experience: A Structured Experience-Management Solution for Self-Evolving Agents under Low-Repetition and Implicit-Reward Environments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05661","citing_title":"Continual Learning Bench: Evaluating Frontier AI Systems in Real-World Stateful Environments","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05008","citing_title":"M$^3$Eval: Multi-Modal Memory Evaluation through Cognitively-Grounded Video Tasks","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04815","citing_title":"Learning While Acting: A Skill-Enhanced Test-Time Co-Evolution Framework for Online Lifelong Learning Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02461","citing_title":"AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07632","citing_title":"Evaluation of ML Resource Utilization Requires Model Life Cycle Assessment","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30306","citing_title":"Always-OnAgents:A Survey of Persistent Memory, State, and Governance in LLMAgents","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27366","citing_title":"MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20625","citing_title":"AlphaMemo: Structured Search-Process Memory for Self-Evolving Alpha Mining Agents","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21463","citing_title":"Mem-$\\pi$: Adaptive Memory through Learning When and What to Generate","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18565","citing_title":"MINTEval: Evaluating Memory under Multi-Target Interference in Long-Horizon Agent Systems","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20857","citing_title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03295","citing_title":"Scaling Teams or Scaling Time? Memory Enabled Lifelong Learning in LLM Multi-Agent Systems","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06365","citing_title":"From Agent Loops to Deterministic Graphs: Execution Lineage for Reproducible AI-Native Work","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08013","citing_title":"Learning CLI Agents with Structured Action Credit under Selective Observation","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15597","citing_title":"LLMs Corrupt Your Documents When You Delegate","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17091","citing_title":"GenericAgent: A Token-Efficient Self-Evolving LLM Agent via Contextual Information Density Maximization (V1.0)","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5","json":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5.json","graph_json":"https://pith.science/api/pith-number/L7Y5VORU66BTY2CDJH5HSMVKE5/graph.json","events_json":"https://pith.science/api/pith-number/L7Y5VORU66BTY2CDJH5HSMVKE5/events.json","paper":"https://pith.science/paper/L7Y5VORU"},"agent_actions":{"view_html":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5","download_json":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5.json","view_paper":"https://pith.science/paper/L7Y5VORU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11942&json=true","fetch_graph":"https://pith.science/api/pith-number/L7Y5VORU66BTY2CDJH5HSMVKE5/graph.json","fetch_events":"https://pith.science/api/pith-number/L7Y5VORU66BTY2CDJH5HSMVKE5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5/action/storage_attestation","attest_author":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5/action/author_attestation","sign_citation":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5/action/citation_signature","submit_replication":"https://pith.science/pith/L7Y5VORU66BTY2CDJH5HSMVKE5/action/replication_record"}},"created_at":"2026-07-05T11:12:31.906223+00:00","updated_at":"2026-07-05T11:12:31.906223+00:00"}