{"as_of":"2026-08-07T19:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ac0bfaf32b53308d763b13422d2104b0f837f3ddcbc9ab81b1c3dddc6f0807fb","coverage":[{"denominator":102,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T01:12:39.559055Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.00155/citation-record","integrity":"/paper/2608.00155/integrity","json":"/paper/2608.00155/citation-record.json","paper":"/paper/2608.00155"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:30.678535Z","title":"A survey of self-evolving agents: What, when, how, and where to evolve on the path to artificial super intelligence.Transactions on Machine Learning Research, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.678535Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4ca9a51d1018489f3314953a54597724a69c1d2e7bd5c8987f54ae9776d4dd1f","observation_id":"67b611d2-1277-424a-96f0-bd6abc117c0c","resolution":{"observed_at":"2026-08-04T01:12:30.678535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:30.740362Z","title":"Position: Agentic evolution is the path to evolving llms.arXiv preprint arXiv:2602.00359, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.740362Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:7d7731b257289401d9f370510b8559901f66ef0c8ca5997490e91ae0bf40fe98","observation_id":"15c3b5f3-0cc4-407d-97dd-53e6f149a015","resolution":{"observed_at":"2026-08-04T01:12:30.740362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.07407","last_updated":"2025-08-31T14:55:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-10T16:07:32Z","title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.07407","snapshot_observed_at":"2026-08-04T01:12:30.844983Z","title":"A comprehensive survey of self-evolving ai agents: A new paradigm bridging foundation models and lifelong agentic systems.arXiv preprint arXiv:2508.07407, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.844983Z"},"links":{"cited_paper":"/paper/2508.07407","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5bbb6d91f92619cb8f2d21cfee55ad2fd536450c48568275dc8404f9cd4a7fd7","observation_id":"774e17c1-81ba-4c06-8184-c3942a218a0c","resolution":{"observed_at":"2026-08-04T01:12:30.844983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:30.945741Z","title":"Agentic context engineering: Evolving contexts for self-improving language models","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:30.945741Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:80fdff0fac1a43a0a3d7721962961195a5b828080fd418a24b13256b9a911e08","observation_id":"f6d0b54b-6b3b-4420-ba41-a3bb96d97d7c","resolution":{"observed_at":"2026-08-04T01:12:30.945741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.30621","last_updated":"2026-05-28T22:16:14Z","snapshot_observed_at":"2026-08-02T11:09:20.796647Z","submitted_at":"2026-05-28T22:16:14Z","title":"Harness Updating Is Not Harness Benefit: Disentangling Evolution Capabilities in Self-Evolving LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.30621","snapshot_observed_at":"2026-08-04T01:12:31.047941Z","title":"Harness updating is not harness benefit: Disentangling evolution capabilities in self-evolving llm agents.arXiv preprint arXiv:2605.30621, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.047941Z"},"links":{"cited_paper":"/paper/2605.30621","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:82ecdc3965a1543010e538bfb34c15d61cfac72b3a54aaafc14d4521c470e70c","observation_id":"4b30595c-b4bf-4321-82f1-c1ac004ba677","resolution":{"observed_at":"2026-08-04T01:12:31.047941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.168633Z","title":"A-mem: Agentic memory for llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.168633Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5f8388874fe172ffd33df2ef357e8f0dd5b93404f143ccdae6a97331ec7fb8d9","observation_id":"4d1cbd62-4880-457e-9d8e-76e4c51ce49d","resolution":{"observed_at":"2026-08-04T01:12:31.168633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03192","last_updated":"2026-02-12T05:43:57Z","snapshot_observed_at":"2026-08-02T21:01:18.325722Z","submitted_at":"2026-01-06T17:14:50Z","title":"MemRL: Self-Evolving Agents via Runtime Reinforcement Learning on Episodic Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03192","snapshot_observed_at":"2026-08-04T01:12:31.238737Z","title":"Memrl: Self-evolving agents via runtime reinforcement learning on episodic memory","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.238737Z"},"links":{"cited_paper":"/paper/2601.03192","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6b696543e87ba7be3685bbd83fc1d85aecbd0b93b53c387f200f5a4170ab3510","observation_id":"5b4a68ac-830d-4563-8af2-bc2c42882100","resolution":{"observed_at":"2026-08-04T01:12:31.238737Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.339997Z","title":"Autoskill: Experience-driven lifelong learning via skill self-evolution.arXiv preprint arXiv:2603.01145, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.339997Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:bbe42c54047f233493b446509d60184dadbae4140463e3b93c9261b8f884e5be","observation_id":"9997acb5-415b-4035-aced-f5f12b482392","resolution":{"observed_at":"2026-08-04T01:12:31.339997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.448963Z","title":"Le, Samira Daruki, Xiangru Tang, Vishy Tirumalashetty, George Lee, Mahsan Rofouei, Hangfei Lin, Jiawei Han, Chen-Yu Lee, and Tomas Pfister","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.448963Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:bb7b860daf4aaa00caf6e44dc65b915ce1a9c8c2ded89dd64a054cc41295a991","observation_id":"2ae5e5f3-dae5-48a1-9f88-e580831d52be","resolution":{"observed_at":"2026-08-04T01:12:31.448963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.553680Z","title":"Memento-skills: Let agents design agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.553680Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:dc630ceb8a336bbb65660cf7c37fbea1b5cc674e35e7413e095aa9185f8dc5a7","observation_id":"669b2ad1-81a2-4075-bd74-1379c5d0b903","resolution":{"observed_at":"2026-08-04T01:12:31.553680Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.25850","last_updated":"2026-05-18T15:11:47Z","snapshot_observed_at":"2026-07-06T23:11:39.318906Z","submitted_at":"2026-04-28T16:55:02Z","title":"Agentic Harness Engineering: Observability-Driven Automatic Evolution of Coding-Agent Harnesses","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.25850","snapshot_observed_at":"2026-08-04T01:12:31.613003Z","title":"Agentic harness engineering: Observability-driven automatic evolution of coding-agent harnesses.arXiv preprint arXiv:2604.25850, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.613003Z"},"links":{"cited_paper":"/paper/2604.25850","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:edf6939dea5238db970b6c96397ea94786b0a0e9fb51cf79ad10c95c203cbf60","observation_id":"aff5ba51-e69a-4fe0-958c-0ca5c67e030e","resolution":{"observed_at":"2026-08-04T01:12:31.613003Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.709243Z","title":"Appworld: A controllable world of apps and people for benchmarking interactive coding agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.709243Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:09dfae61ded526c9f5deb30dfcfc224041b765871c73095f9858f4da77cc80f2","observation_id":"8650ca55-dd31-4e2b-b6c2-2e8233c3b8e2","resolution":{"observed_at":"2026-08-04T01:12:31.709243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.826496Z","title":"Gonzalez","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.826496Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:90e76b7e4b12ab9cf45d76cd72f0972d3197a37712e57af63199c3070b47d2dc","observation_id":"0313bafd-50e0-4c88-95b5-24fa3f54c936","resolution":{"observed_at":"2026-08-04T01:12:31.826496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:31.918060Z","title":"Swe-bench: Can language models resolve real-world github issues? In Proc","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.918060Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:2179ad0e364b0655afdeb0426902bf6bc8aade81e0dbd181d11b865b60e74ac1","observation_id":"192ac56d-3eb2-4e4d-966f-b596268400d9","resolution":{"observed_at":"2026-08-04T01:12:31.918060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.14249","snapshot_observed_at":"2026-08-04T01:12:31.996592Z","title":"Humanity’s last exam.arXiv preprint arXiv:2501.14249, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:31.996592Z"},"links":{"cited_paper":"/paper/2501.14249","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8514ab12b2ef43e50610fbccb4db869c566b5225efc477cba6ea852663abb795","observation_id":"c22fc5a9-a231-4699-b27a-21e74d18ae93","resolution":{"observed_at":"2026-08-04T01:12:31.996592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-08-04T01:12:32.087301Z","title":"τ 2- Bench: Evaluating Conversational Agents in a Dual-Control Environment.arXiv preprint arXiv:2506.07982, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.087301Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:46b3303b29742281efc58d11c8427323d8295eef4787efa5794c6f3f87eaa088","observation_id":"45ed4434-2e42-4aac-a302-79bb5630afb5","resolution":{"observed_at":"2026-08-04T01:12:32.087301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.156197Z","title":"Stream- bench: Towards benchmarking continuous improvement of language agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.156197Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d65f455ee4481077f2fdacedf0fe430882a559f6175cebfec42d0c16840e4f20","observation_id":"0ab7ed43-9bc6-413b-8b5e-082f746f58f4","resolution":{"observed_at":"2026-08-04T01:12:32.156197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.20857","last_updated":"2026-05-18T16:18:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-25T21:08:07Z","title":"Evo-Memory: Benchmarking LLM Agent Test-time Learning with Self-Evolving Memory","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.20857","snapshot_observed_at":"2026-08-04T01:12:32.201860Z","title":"Chi, Chi Wang, Shuo Chen, Fernando Pereira, Wang-Cheng Kang, and Derek Zhiyuan Cheng","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.201860Z"},"links":{"cited_paper":"/paper/2511.20857","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:cbe4c68c7edb923db0e0864249dd2e99848fb91b55b275b60a06e36590e6de85","observation_id":"19de0ee3-c2cd-4dac-84f9-1710b859a36c","resolution":{"observed_at":"2026-08-04T01:12:32.201860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-04T01:12:32.267699Z","title":"Openai gpt-5 system card.arXiv preprint arXiv:2601.03267, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.267699Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:357d386d0af9527e20928e2ae83bf0fa68512d4afc83b08cccea60a41fca7df4","observation_id":"b8cc89d3-74d9-495a-a8d0-64d85f8c11ec","resolution":{"observed_at":"2026-08-04T01:12:32.267699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.359163Z","title":"Gemini 3.1 Pro model card, February 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.359163Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a82b1181ed994047aa22cdbe695eb286a48631435ef56539968202d75e3d3d64","observation_id":"5ba34156-2f11-4e2f-be16-4be50f196bea","resolution":{"observed_at":"2026-08-04T01:12:32.359163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.441241Z","title":"Introducing Claude Opus 4.7, April 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.441241Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:655980457983fe1fc259ecaccd082d461779b6ff90ec55adb723891fbdcbaaf0","observation_id":"b9ba58dd-a90c-4480-9b54-3e6fb70502a1","resolution":{"observed_at":"2026-08-04T01:12:32.441241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.517774Z","title":"Browsecomp-plus: A more fair and transparent evaluation benchmark of deep-research agent","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.517774Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8787cea5a936a1447044ebf419348840a679737242c9c3eb89ad68a1ee46e520","observation_id":"1e4e000b-6c7e-4e9c-b814-b4951cdeda99","resolution":{"observed_at":"2026-08-04T01:12:32.517774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.557621Z","title":"Test-time training with self-supervision for generalization under distribution shifts","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.557621Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c86c834e49b76cfdfc6f21499a1cb5518e1c980e3dfb2605fd5e0f01f96c97a5","observation_id":"a9c9edd8-577d-471a-81e4-899bf8f7293c","resolution":{"observed_at":"2026-08-04T01:12:32.557621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.616665Z","title":"Do we really need to access the source data? source hypothesis transfer for unsupervised domain adaptation","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.616665Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:451818a0bad97f9603744ce177a724acbe73ce992cd89dbc8ef48375a59d34c6","observation_id":"1ef7ec98-9596-452d-b766-b5a37395abac","resolution":{"observed_at":"2026-08-04T01:12:32.616665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.706158Z","title":"Overcoming catastrophic forgetting in neural networks.Proceedings of the national academy of sciences, 114(13):3521–3526, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.706158Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:1da800d394da30396003db392a3c4a9a5cfd3ba7e30ae54150571981accaefda","observation_id":"4ec2d7e4-606f-45df-bcbf-08412cdd5a1a","resolution":{"observed_at":"2026-08-04T01:12:32.706158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.785239Z","title":"Gradient episodic memory for continual learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.785239Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c766832613d4b9535929af4d36ebc697ebb02fe593f5424e47fedec04e7b22d8","observation_id":"7128a647-c5e2-4fc7-bee4-44afb379adaa","resolution":{"observed_at":"2026-08-04T01:12:32.785239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.834609Z","title":"Scaling llm test-time compute optimally can be more effective than scaling model parameters","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.834609Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:99705ce50e5e1ff1e5e10373609f96894b1d81b1c1d195f41da7fbb031ca1cca","observation_id":"13b29ec8-639f-4f57-9e71-ec2d6c5b41ed","resolution":{"observed_at":"2026-08-04T01:12:32.834609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.907311Z","title":"Test-time training on nearest neighbors for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.907311Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:530d8b6ce1e865f158ac7699e756b41a467404707ca9694ff6bc40f89bde6335","observation_id":"ea24b3aa-c55c-429d-8f36-4f6fd65c80e7","resolution":{"observed_at":"2026-08-04T01:12:32.907311Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.947220Z","title":"Efficiently learning at test-time: Active fine-tuning of llms","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.947220Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:2ea5439e49317326824aa0df7281899d052b2de4fab51644bfbb6f9e6bc237d9","observation_id":"69b76d30-5c03-4755-a817-87bedd5539bf","resolution":{"observed_at":"2026-08-04T01:12:32.947220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:32.990733Z","title":"The surprising effectiveness of test-time training for few-shot learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:32.990733Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:13e7c906feb08c510a8eeef7cd144f6033b17f96d8c83e2c288c456dc020a797","observation_id":"8353d4e8-7813-4986-b32e-8b3fd29c1545","resolution":{"observed_at":"2026-08-04T01:12:32.990733Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.052856Z","title":"In-place test-time training","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.052856Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a0ec99f502a7c2e68d024ee0d878b3956b42f015e4e8f02f8829a31942cd0ec6","observation_id":"1aaae17b-f559-45b8-b4aa-042734c11429","resolution":{"observed_at":"2026-08-04T01:12:33.052856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.147161Z","title":"Test-time adaptation for llm agents via environment interaction","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.147161Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:eeb3e4952fbc372f4bac75524e99d9bb85ffd718c798b4870bf5075da6ca04fe","observation_id":"78bb23fd-0340-46e7-a979-5c9cb6c60f57","resolution":{"observed_at":"2026-08-04T01:12:33.147161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.266388Z","title":"Test-time learning for large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.266388Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:0464028a85b9dacd3e1d3e2f7644b48e9bb3d9664c3b8625a2edee36eae3344a","observation_id":"8c6d34b6-56e8-42fd-be9c-40b104d6fb5b","resolution":{"observed_at":"2026-08-04T01:12:33.266388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.347224Z","title":"Ttrl: Test-time reinforcement learning","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.347224Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:41aacb26c3c7b588d92c467068b743601c53fbc6f1c95e41ea72ebc43cca1540","observation_id":"31c98bfe-fc30-4da1-b482-26138dfd8ce1","resolution":{"observed_at":"2026-08-04T01:12:33.347224Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.407176Z","title":"Learn- ing on the job: Test-time curricula for targeted reinforcement learning.arXiv preprint arXiv:2510.04786, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.407176Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:eed322b5abbf6e81c6e01b2ddba9d89cc8ff63db0473a116508327c54fc99881","observation_id":"560c579e-c5e5-4904-893b-432431fa2cb1","resolution":{"observed_at":"2026-08-04T01:12:33.407176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.486960Z","title":"Learning to discover at test time","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.486960Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:817562b74f59dbf1d87689dc27b3ff2e22f59d82525aff69d89962945c1713f2","observation_id":"729ca982-0b5d-48a6-bc45-3aa1aa448f4f","resolution":{"observed_at":"2026-08-04T01:12:33.486960Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.584703Z","title":"Collaborative multi-agent test-time reinforcement learning for reasoning.arXiv preprint arXiv:2601.09667, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.584703Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:03a4ffe7dcde0c1a6820dfaa2b4be18e788b064fb6305d3751e71e8d2c980295","observation_id":"0dbd5315-bd55-49ce-a095-a6213b44dc44","resolution":{"observed_at":"2026-08-04T01:12:33.584703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.672842Z","title":"What if consensus lies? selective-complementary reinforcement learning at test time","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.672842Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6792610fd211ab768a2686c37aa2f59cc486558da455d27c128e2afd525efc66","observation_id":"8a992a21-8df4-4314-91b8-8d6506facb9c","resolution":{"observed_at":"2026-08-04T01:12:33.672842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.783687Z","title":"Ttsr: Test-time self-reflection for continual reasoning improvement.arXiv preprint arXiv:2603.03297, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.783687Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d2c18c9fe59bd81bb09bbddc6f1d9445634e7c88cde5997aa7e32a6ed261a98b","observation_id":"0cb42fec-7ea7-47d8-add2-4a5c603aa6c1","resolution":{"observed_at":"2026-08-04T01:12:33.783687Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:33.899505Z","title":"Ttcs: Test-time curriculum synthesis for self-evolving.arXiv preprint arXiv:2601.22628, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:33.899505Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:13058a7a5a5cb672bd0fec9179cfcc2d50cd54b08b64038df2bc4e3e6df9666f","observation_id":"d76d5b23-7867-48c6-ba76-c7b3adbf5fd3","resolution":{"observed_at":"2026-08-04T01:12:33.899505Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.14477","last_updated":"2026-07-14T21:30:34Z","snapshot_observed_at":"2026-08-02T14:06:39.249902Z","submitted_at":"2026-05-14T07:18:12Z","title":"Test-Time Learning with an Evolving Library","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.14477","snapshot_observed_at":"2026-08-04T01:12:34.009288Z","title":"Test-time learning with an evolving library.arXiv preprint arXiv:2605.14477, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.009288Z"},"links":{"cited_paper":"/paper/2605.14477","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a77414cecde7ed1c3cef970a4ff90267d96966be8c7138b8221c34232b62fdd4","observation_id":"b01a108e-71ff-4206-a9bc-190f1a4e571e","resolution":{"observed_at":"2026-08-04T01:12:34.009288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.16986","last_updated":"2026-07-29T10:19:54Z","snapshot_observed_at":"2026-08-02T13:52:40.874253Z","submitted_at":"2026-05-16T13:14:15Z","title":"Skills on the Fly: Test-Time Adaptive Skill Synthesis for LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.16986","snapshot_observed_at":"2026-08-04T01:12:34.114246Z","title":"Skills on the fly: Test-time adaptive skill synthesis for llm agents.arXiv preprint arXiv:2605.16986, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.114246Z"},"links":{"cited_paper":"/paper/2605.16986","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:e8568e97b8690b7285e9b6eecc1ed9aeb90fc518d6768ec0d35a77e896cc801f","observation_id":"eb9d85b7-402a-49fc-aec2-7efb0b250981","resolution":{"observed_at":"2026-08-04T01:12:34.114246Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.226463Z","title":"Tarse: Test-time adaptation via retrieval of skills and experience for reasoning agents.arXiv preprint arXiv:2603.01241, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.226463Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5cae0917cc4b13b9067b62bee8214a6c98a7e84bd9c4b5aeb4c793c4707660aa","observation_id":"4b2262f2-6162-4fac-abde-a3da7f3f3208","resolution":{"observed_at":"2026-08-04T01:12:34.226463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.337320Z","title":"Agentic plan caching: Test-time memory for fast and cost-efficient llm agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.337320Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:946869aaeeeddf5ffc8e46008d3b0c04771d6af2a9bf8494741ef6acd3d34336","observation_id":"26fb1c9a-50dd-4827-84ac-1aa5afaa48dd","resolution":{"observed_at":"2026-08-04T01:12:34.337320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.03224","last_updated":"2026-06-06T09:48:09Z","snapshot_observed_at":"2026-08-06T23:53:57.800987Z","submitted_at":"2026-02-03T07:52:26Z","title":"TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.03224","snapshot_observed_at":"2026-08-04T01:12:34.442077Z","title":"Tame: A trustworthy test-time evolution of agent memory with systematic benchmarking.arXiv preprint arXiv:2602.03224, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.442077Z"},"links":{"cited_paper":"/paper/2602.03224","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4b67e9b9017cd10e75425244014730dfd6eb0c345e05cc4a4dcd981ae82f3421","observation_id":"ed2c9bfc-f745-4cf6-b322-a65549161db3","resolution":{"observed_at":"2026-08-04T01:12:34.442077Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.540281Z","title":"Self-improving llm agents at test-time.arXiv preprint arXiv:2510.07841, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.540281Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:4dcd10e3e775933c22a0b6771d9398e494313cafbc51710abb6a3673f8b4c446","observation_id":"18107843-204b-4eb5-b983-e6c2028238e2","resolution":{"observed_at":"2026-08-04T01:12:34.540281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.652568Z","title":"Just- in-time reinforcement learning: Continual learning in llm agents without gradient updates","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.652568Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:604bf5039f320f736bc7f62362a2a0bba70052d3c8d578b6d99a710a6b41b2d3","observation_id":"b09b02a8-d8b8-453a-b705-88ffabf8a7c3","resolution":{"observed_at":"2026-08-04T01:12:34.652568Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.761418Z","title":"Panini: Continual learning in token space via structured memory","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.761418Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ae5bb2c77dc9eca9902aa4a75043e5d9cae4fed6984825305d253fd2cdc252d3","observation_id":"9f5ea4cc-4e98-4228-9cbd-eb3796899c71","resolution":{"observed_at":"2026-08-04T01:12:34.761418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03641","last_updated":"2026-04-11T14:07:11Z","snapshot_observed_at":"2026-07-06T22:40:56.386532Z","submitted_at":"2026-01-07T06:43:50Z","title":"Agent-Dice: Disentangling Knowledge Updates via Geometric Consensus for Agent Continual Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2601.03641","snapshot_observed_at":"2026-08-04T01:12:34.866944Z","title":"Agent-dice: Disentangling knowledge updates via geometric consensus for agent continual learning.arXiv preprint arXiv:2601.03641, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.866944Z"},"links":{"cited_paper":"/paper/2601.03641","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:612e365431a2859092feeb16d7b6c80ce8105309fc60d557277b5c54b0303311","observation_id":"5a6a2359-872e-42b0-bcc8-368eb0a6533b","resolution":{"observed_at":"2026-08-04T01:12:34.866944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:34.939316Z","title":"Mssr: Memory-aware adaptive replay for continual llm fine-tuning.arXiv preprint arXiv:2603.09892, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:34.939316Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:aaa3c09835e98c5eed9c20b2025f4a420fb77ff78fa308dac41273cdaece4633","observation_id":"df12cecf-038a-4dc1-8a2b-c5be36a6d665","resolution":{"observed_at":"2026-08-04T01:12:34.939316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:35.017911Z","title":"Learning to continually learn via meta-learning agentic memory designs.arXiv preprint arXiv:2602.07755, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.017911Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:caf73ee28960185f5ac1fe07b0b0902f8555c8c82e5a76f9045ff49dff41689e","observation_id":"fec8e04f-7212-49fe-a4ed-8b154e68aa71","resolution":{"observed_at":"2026-08-04T01:12:35.017911Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:35.087726Z","title":"Xskill: Continual learning from experience and skills in multimodal agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.087726Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ab064cd0908af6f43d8d68acb1fcf6f43216ac5c4bad1a207ebebad68404c8da","observation_id":"fb473b65-f10e-480b-bb9b-04a615958777","resolution":{"observed_at":"2026-08-04T01:12:35.087726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.16856","last_updated":"2026-06-29T12:13:29Z","snapshot_observed_at":"2026-07-13T23:28:27.562632Z","submitted_at":"2026-03-17T17:57:49Z","title":"Online Experiential Learning for Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.16856","snapshot_observed_at":"2026-08-04T01:12:35.159727Z","title":"Online experiential learning for language models.arXiv preprint arXiv:2603.16856, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.159727Z"},"links":{"cited_paper":"/paper/2603.16856","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:5ca0783dd68a2ea890879c1aa9e7bdfafb7a2a0fe2e4981cd3476ca5f5d6c048","observation_id":"da43f976-c6ce-463c-9799-361ce994f6aa","resolution":{"observed_at":"2026-08-04T01:12:35.159727Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:35.227678Z","title":"Adaptive collaboration with humans: Metacognitive policy optimization for multi-agent llms with continual learning","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.227678Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ebe53c2713473874bb954e89e0652f117ae2e08a82f9fbd9000fe3d712276abf","observation_id":"ddc78d07-c649-491c-9b06-7196fe1fb0e7","resolution":{"observed_at":"2026-08-04T01:12:35.227678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02474","last_updated":"2026-05-24T19:01:09Z","snapshot_observed_at":"2026-08-03T05:25:38.558727Z","submitted_at":"2026-02-02T18:53:28Z","title":"MemSkill: Learning and Evolving Memory Skills for Self-Evolving Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.02474","snapshot_observed_at":"2026-08-04T01:12:35.336093Z","title":"Memskill: Learning and evolving memory skills for self-evolving agents.arXiv preprint arXiv:2602.02474, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.336093Z"},"links":{"cited_paper":"/paper/2602.02474","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:e7d7943ebeea60ad306ff864165d76f97e6636005859f76f9e382ded46ebaa6b","observation_id":"d42845b4-b90e-46aa-b1b5-2dbf86837931","resolution":{"observed_at":"2026-08-04T01:12:35.336093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19413","last_updated":"2025-04-28T01:46:35Z","snapshot_observed_at":"2026-08-02T07:32:11.339534Z","submitted_at":"2025-04-28T01:46:35Z","title":"Mem0: Building Production-Ready AI Agents with Scalable Long-Term Memory","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.19413","snapshot_observed_at":"2026-08-04T01:12:35.459460Z","title":"Mem0: Building production-ready ai agents with scalable long-term memory.arXiv preprint arXiv:2504.19413, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.459460Z"},"links":{"cited_paper":"/paper/2504.19413","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:61d59d14e6c2a9b9c4a4849e9475682b68f495e7c22b5a37a03a19a0d9ab2b05","observation_id":"227b17f5-290a-48c0-94bd-9da4a8674c98","resolution":{"observed_at":"2026-08-04T01:12:35.459460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.08234","last_updated":"2026-02-09T03:17:17Z","snapshot_observed_at":"2026-07-06T22:45:05.823859Z","submitted_at":"2026-02-09T03:17:17Z","title":"SkillRL: Evolving Agents via Recursive Skill-Augmented Reinforcement Learning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.08234","snapshot_observed_at":"2026-08-04T01:12:35.579166Z","title":"Skillrl: Evolving agents via recursive skill-augmented reinforcement learning.arXiv preprint arXiv:2602.08234, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.579166Z"},"links":{"cited_paper":"/paper/2602.08234","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ffd950b83c261591498d3e0150f4b57d555ff27ecfde28956a365be0967bd359","observation_id":"e6271b35-7585-4cff-ac8d-17152f2132e8","resolution":{"observed_at":"2026-08-04T01:12:35.579166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.06614","last_updated":"2026-05-07T17:31:50Z","snapshot_observed_at":"2026-07-06T23:19:05.764765Z","submitted_at":"2026-05-07T17:31:50Z","title":"SkillOS: Learning Skill Curation for Self-Evolving Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.06614","snapshot_observed_at":"2026-08-04T01:12:35.690973Z","title":"Skillos: Learning skill curation for self-evolving agents.arXiv preprint arXiv:2605.06614, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.690973Z"},"links":{"cited_paper":"/paper/2605.06614","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:336a127cdacbe496bd3b0c832c8a24e178c8d398fbbf86e32ddc08e449518746","observation_id":"1fcce961-c4c5-4a04-8cba-613fc43b0b43","resolution":{"observed_at":"2026-08-04T01:12:35.690973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.02766","last_updated":"2026-03-03T09:07:22Z","snapshot_observed_at":"2026-08-04T19:32:12.955430Z","submitted_at":"2026-03-03T09:07:22Z","title":"EvoSkill: Automated Skill Discovery for Multi-Agent Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.02766","snapshot_observed_at":"2026-08-04T01:12:35.780877Z","title":"Evoskill: Automated skill discovery for multi-agent systems.arXiv preprint arXiv:2603.02766, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.780877Z"},"links":{"cited_paper":"/paper/2603.02766","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f583af0430ea5804deaeaa66bf83d499f1f8de5f3f55f4066b7fe05bdff6693b","observation_id":"60d83db8-2aa5-4635-b235-d2f7d73e311e","resolution":{"observed_at":"2026-08-04T01:12:35.780877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.06741","last_updated":"2026-06-04T21:55:48Z","snapshot_observed_at":"2026-07-06T23:46:28.942186Z","submitted_at":"2026-06-04T21:55:48Z","title":"OpenSkill: Open-World Self-Evolution for LLM Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.06741","snapshot_observed_at":"2026-08-04T01:12:35.900154Z","title":"Yu, Ran Xu, Xiang Li, and Lichao Sun","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:35.900154Z"},"links":{"cited_paper":"/paper/2606.06741","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:cea24f2502646202162e847081a83882e0c675a0ae2e59a5b42cf9b4aea57886","observation_id":"77a4cd23-7cf0-42d9-bfbc-c364071b1d2d","resolution":{"observed_at":"2026-08-04T01:12:35.900154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.23904","last_updated":"2026-05-25T17:58:16Z","snapshot_observed_at":"2026-08-05T06:54:08.846880Z","submitted_at":"2026-05-22T17:59:50Z","title":"SkillOpt: Executive Strategy for Self-Evolving Agent Skills","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.23904","snapshot_observed_at":"2026-08-04T01:12:36.013114Z","title":"Skillopt: Executive strategy for self-evolving agent skills.arXiv preprint arXiv:2605.23904, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.013114Z"},"links":{"cited_paper":"/paper/2605.23904","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:10640dcd1acc3c52d6f9376ea66fb72465829db76130fbce765c33ba1dadf2a9","observation_id":"3406787a-4e71-4b33-9ab7-b8a54d9deec6","resolution":{"observed_at":"2026-08-04T01:12:36.013114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.28052","last_updated":"2026-03-30T05:33:50Z","snapshot_observed_at":"2026-07-06T22:51:00.579062Z","submitted_at":"2026-03-30T05:33:50Z","title":"Meta-Harness: End-to-End Optimization of Model Harnesses","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.28052","snapshot_observed_at":"2026-08-04T01:12:36.133344Z","title":"Meta-harness: End-to-end optimization of model harnesses.arXiv preprint arXiv:2603.28052, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.133344Z"},"links":{"cited_paper":"/paper/2603.28052","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ee2c125328b45c5f87c2c517d92f18a6b0a0614cf0604062d02726e799bf3af7","observation_id":"9f8206b5-8e87-4d2a-82ea-3561f5a75200","resolution":{"observed_at":"2026-08-04T01:12:36.133344Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.247731Z","title":"Evoconfig: Self-evolving multi-agent systems for efficient autonomous environment configuration.arXiv preprint arXiv:2601.16489, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.247731Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:20e0a2b6256f9ed2e577510d5451ee85515b7a41515ff4f78c921c9cfe777bca","observation_id":"f4deb2ae-ff8d-44e6-a47e-4bf4271f2963","resolution":{"observed_at":"2026-08-04T01:12:36.247731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.328391Z","title":"Selaur: Self evolving llm agent via uncertainty-aware rewards","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.328391Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:16a0e594977e2d904b5d532d84f90ebd67fab1ba2f8dae455ba255cf2be5eda2","observation_id":"fd65faba-7aaf-4848-ae01-57b277ee676c","resolution":{"observed_at":"2026-08-04T01:12:36.328391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.28814","last_updated":"2026-05-27T17:59:15Z","snapshot_observed_at":"2026-07-06T23:38:19.096349Z","submitted_at":"2026-05-27T17:59:15Z","title":"Self-Improving Language Models with Bidirectional Evolutionary Search","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.28814","snapshot_observed_at":"2026-08-04T01:12:36.464200Z","title":"Kakade, and Yilun Du","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.464200Z"},"links":{"cited_paper":"/paper/2605.28814","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c4299f4c6ce1196d04ebb19f6edd18ea8eb036538fd41027cbf604ff9e613e36","observation_id":"78298233-2985-415e-a37a-9f4e5254b1ca","resolution":{"observed_at":"2026-08-04T01:12:36.464200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.602393Z","title":"Tool-r0: Self-evolving llm agents for tool-learning from zero data.arXiv preprint arXiv:2602.21320, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.602393Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8d87836d1a42473858aec3101d9fd95e3f20a1418fbdced89e5831472bd5e19f","observation_id":"42b75e4e-19e0-460c-86e8-ec6acc6fd764","resolution":{"observed_at":"2026-08-04T01:12:36.602393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.737392Z","title":"Rein- forcing chain-of-thought reasoning with self-evolving rubrics.arXiv preprint arXiv:2602.10885, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.737392Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:de0826d6d712b04683c416fbe1e7e82ff368a7cebb3832250f8c81b12e53f08b","observation_id":"c2e9eb84-99d0-47dc-beda-b6654d619d6a","resolution":{"observed_at":"2026-08-04T01:12:36.737392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:36.901016Z","title":"Metagen: Self-evolving roles and topologies for multi-agent llm reasoning.arXiv preprint arXiv:2601.19290, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:36.901016Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:47e03a79816fb9b7af9f1c17be9d74df9e7df5a02aab446b15cd5ee7a148a400","observation_id":"8d4c3a23-e484-4c81-82a6-2206d90e5e79","resolution":{"observed_at":"2026-08-04T01:12:36.901016Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.006298Z","title":"Self-evolving multi-agent collaboration networks for software development","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.006298Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6675431add820f3d88dfa13d0efa3979a448807d82360368ce1e5051f4911cfb","observation_id":"81c2eb88-de5c-4ea8-b3e8-4ef04ac48b49","resolution":{"observed_at":"2026-08-04T01:12:37.006298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18646","last_updated":"2026-04-14T11:23:32Z","snapshot_observed_at":"2026-08-02T05:01:07.687054Z","submitted_at":"2025-05-24T11:12:14Z","title":"SEW: Self-Evolving Agentic Workflows for Automated Code Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.18646","snapshot_observed_at":"2026-08-04T01:12:37.085301Z","title":"Sew: Self-evolving agentic workflows for automated code generation.arXiv preprint arXiv:2505.18646, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.085301Z"},"links":{"cited_paper":"/paper/2505.18646","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:097557dcae62340ad9407ae2c2b2efea342f29cd17fdb92a3f9c9ea43c8e6dc2","observation_id":"fb119d41-1db2-491c-b863-2c7cd2db0aad","resolution":{"observed_at":"2026-08-04T01:12:37.085301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.179134Z","title":"Evotool: Self-evolving tool-use policy optimization in llm agents via blame-aware mutation and diversity-aware selection.arXiv preprint arXiv:2603.04900, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.179134Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:6f1057872a7f7bf423f4e6be81bb9c20a97fd87cf0a24e0be711a3ea20ef9f77","observation_id":"a438e6fe-bc99-4020-aadc-cd4a1d5f5693","resolution":{"observed_at":"2026-08-04T01:12:37.179134Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13131","last_updated":"2025-06-16T06:37:18Z","snapshot_observed_at":"2026-08-07T04:21:43.190472Z","submitted_at":"2025-06-16T06:37:18Z","title":"AlphaEvolve: A coding agent for scientific and algorithmic discovery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13131","snapshot_observed_at":"2026-08-04T01:12:37.248349Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.248349Z"},"links":{"cited_paper":"/paper/2506.13131","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:ce5aacdf4f83fd49a50608d916a210c906f36b0789ba3059c3bdf9a713bf74b9","observation_id":"3ecd6a5f-8ebc-4011-8b8c-1248324a388d","resolution":{"observed_at":"2026-08-04T01:12:37.248349Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.318781Z","title":"Evotest: Evolutionary test-time learning for self-improving agentic systems","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.318781Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:41cb92618ac5f7dd0bc7e5b9e72b1747d5ef091a30826ba73e9763c992e7d175","observation_id":"6a19d7cd-4a82-425a-8a93-a521749a1200","resolution":{"observed_at":"2026-08-04T01:12:37.318781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.425048Z","title":"Building self-evolving agents via experience-driven lifelong learning: A framework and benchmark.arXiv preprint arXiv:2508.19005, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.425048Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:03085b421007d4b5d65f137cd35b30bbdba54e7186f1194389ba230e91cd9ee9","observation_id":"3a3b0b20-44fa-43b8-a733-cfb5ed07a0be","resolution":{"observed_at":"2026-08-04T01:12:37.425048Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.498006Z","title":"Optimizing generative ai by backpropagating language model feedback.Nature, 639:609–616, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.498006Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f0cdf6802a5c9a75a7789aec74b05c00dc7f59853dac37fd9f53980290548f98","observation_id":"b76f30c6-aa60-4f0a-8914-2712ff2475be","resolution":{"observed_at":"2026-08-04T01:12:37.498006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.591245Z","title":"Your agent may misevolve: Emergent risks in self-evolving llm agents","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.591245Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:0ba81735cfe3ea576abcb16576d9a97bf297d3c006a0ad7220823be57935d0b8","observation_id":"73fce81a-b1bf-4152-9555-ee4a82465c9c","resolution":{"observed_at":"2026-08-04T01:12:37.591245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.698493Z","title":"Webshop: Towards scalable real-world web interaction with grounded language agents","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.698493Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:29ffacc4884c87692c17b336129c2876727bddf50472c40b1313d572fe6f0330","observation_id":"d6b8e116-6b94-4a8e-8c9e-eafca8f43605","resolution":{"observed_at":"2026-08-04T01:12:37.698493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.811240Z","title":"Xu, Hao Zhu, Xuhui Zhou, Robert Lo, Abishek Sridhar, Xianyi Cheng, Tianyue Ou, Yonatan Bisk, Daniel Fried, Uri Alon, and Graham Neubig","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.811240Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:eed024241cfcb1f06b81af1a1e6eb9daa1ea3077d84de463c1bf7aa8c1b3a32c","observation_id":"2935ea0f-764a-4a2a-8341-5d42cd473aad","resolution":{"observed_at":"2026-08-04T01:12:37.811240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.916387Z","title":"Webvoyager: Building an end-to-end web agent with large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.916387Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:cbc02994a9871d5a9c3b86a74cc07234508c83ab51f566c6f4bb1b5f54f1675d","observation_id":"188f85ae-edff-4a3f-aa09-91fc9df16cc3","resolution":{"observed_at":"2026-08-04T01:12:37.916387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:37.996496Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:37.996496Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:88f7bd0faec05aae7360ce4f4d16f0595af5e46bdb168bcf9a363c50d4b48655","observation_id":"bf78fd0c-0c32-48c7-86ad-fc0033c497ef","resolution":{"observed_at":"2026-08-04T01:12:37.996496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.00933","last_updated":"2026-05-19T23:26:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-31T23:19:39Z","title":"MCP-Atlas: A Large-Scale Benchmark for Tool-Use Competency with Real MCP Servers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.00933","snapshot_observed_at":"2026-08-04T01:12:38.075464Z","title":"Mcp-atlas: A large-scale benchmark for tool-use competency with real mcp servers.arXiv preprint arXiv:2602.00933, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.075464Z"},"links":{"cited_paper":"/paper/2602.00933","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:2d02530ede4bdf3935c1628f70d210c0d01a5e0c446b9cc42b5acf5ff61a480c","observation_id":"592bab4f-14f5-43de-8234-c160ebd7d284","resolution":{"observed_at":"2026-08-04T01:12:38.075464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.158997Z","title":"The tool decathlon: Benchmarking language agents for diverse, realistic, and long-horizon task execution","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.158997Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:678b7217c86ca9dbce536526ed0e96956042c67dbc88b1b7a274f1689dd13675","observation_id":"f05cad0a-6ce5-48e5-82e9-5d21a17a8a02","resolution":{"observed_at":"2026-08-04T01:12:38.158997Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.214251Z","title":"Terminal-bench: Benchmarking agents on hard, realistic tasks in command line interfaces","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.214251Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:3e4ca0545e8d071c40f240c1b89532215f99c30a12a2cbb7ce77656a5f9a30a5","observation_id":"02c2e747-2463-4d6e-b47b-e9c0e5f07032","resolution":{"observed_at":"2026-08-04T01:12:38.214251Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.287770Z","title":"Cybergym: Evaluating AI agents’ real-world cybersecurity capabilities at scale","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.287770Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:e59affdf11fd29713835f09b46f2fffca37c00a4e208673ee60a398e772ef683","observation_id":"4cc7197c-1adf-4548-8a30-f846e95381f2","resolution":{"observed_at":"2026-08-04T01:12:38.287770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2602.12670","last_updated":"2026-03-13T07:33:01Z","snapshot_observed_at":"2026-07-06T22:45:44.135338Z","submitted_at":"2026-02-13T07:06:06Z","title":"SkillsBench: Benchmarking How Well Agent Skills Work Across Diverse Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2602.12670","snapshot_observed_at":"2026-08-04T01:12:38.372036Z","title":"Skillsbench: Benchmarking how well agent skills work across diverse tasks.arXiv preprint arXiv:2602.12670, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.372036Z"},"links":{"cited_paper":"/paper/2602.12670","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8facdd4aa40a06c30508a3db69a3c28008e6c05bbb01cf92ea2cad5767863880","observation_id":"81771bf5-edef-4fed-94ad-808de9210a98","resolution":{"observed_at":"2026-08-04T01:12:38.372036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.17308","last_updated":"2026-04-19T07:51:46Z","snapshot_observed_at":"2026-07-06T23:04:28.260463Z","submitted_at":"2026-04-19T07:51:46Z","title":"SkillFlow:Benchmarking Lifelong Skill Discovery and Evolution for Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.17308","snapshot_observed_at":"2026-08-04T01:12:38.455621Z","title":"Skillflow: Benchmarking lifelong skill discovery and evolution for autonomous agents.arXiv preprint arXiv:2604.17308, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.455621Z"},"links":{"cited_paper":"/paper/2604.17308","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:d4c590c846f4144a3c815e6d1b71b3b57e9182c503c7f5560458d9ea65968916","observation_id":"b6e25965-c2c8-4b22-b36c-349ab6a533c4","resolution":{"observed_at":"2026-08-04T01:12:38.455621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.536812Z","title":"Gaia: a benchmark for general ai assistants","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.536812Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:81a0e135fa992f1b3c800a72a77772c004e8a9559ce46d068900614aae78aab2","observation_id":"b47a9d7b-74e2-4fe3-8fbd-963edfbf6074","resolution":{"observed_at":"2026-08-04T01:12:38.536812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-03T00:43:33.338074Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-04T01:12:38.625544Z","title":"Browsecomp: A simple yet challenging benchmark for browsing agents.arXiv preprint arXiv:2504.12516, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.625544Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:fd1ab6801b954d9d210fa27959a5a6565f52b1a7fad7ed23bf897dde29e25c45","observation_id":"b512849c-9191-4ddc-a38b-e8badd0f1281","resolution":{"observed_at":"2026-08-04T01:12:38.625544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.04374","last_updated":"2025-10-05T21:36:43Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T21:36:43Z","title":"GDPval: Evaluating AI Model Performance on Real-World Economically Valuable Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.04374","snapshot_observed_at":"2026-08-04T01:12:38.705418Z","title":"Gdpval: Evaluating ai model performance on real-world economically valuable tasks.arXiv preprint arXiv:2510.04374, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.705418Z"},"links":{"cited_paper":"/paper/2510.04374","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f07125d1e61726e9e55e975a71aa7ac3c704af2d8fb92e3cf508533befd63d4b","observation_id":"6f0dc2de-6651-456f-a1aa-1c902b2c9a9f","resolution":{"observed_at":"2026-08-04T01:12:38.705418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.00828","last_updated":"2025-05-20T18:22:10Z","snapshot_observed_at":"2026-08-07T15:30:31.121575Z","submitted_at":"2025-05-20T18:22:10Z","title":"Finance Agent Benchmark: Benchmarking LLMs on Real-world Financial Research Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.00828","snapshot_observed_at":"2026-08-04T01:12:38.779498Z","title":"Finance agent benchmark: Benchmarking llms on real-world financial research tasks.arXiv preprint arXiv:2508.00828, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.779498Z"},"links":{"cited_paper":"/paper/2508.00828","citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:1d6412b6fd1a8a0b08307040c80eeb2dc68205e1b40597c450f9d4615b91bb42","observation_id":"9f015896-0a35-463c-a30a-23394db146bc","resolution":{"observed_at":"2026-08-04T01:12:38.779498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.857487Z","title":"Agent workflow memory","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.857487Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:cfecd0fbbf9325fb7626cc7a4e865a4519504e33a4eac3eb7c932d6626eb19b7","observation_id":"ba0c0772-e8a4-4962-8322-b8fd75fcc637","resolution":{"observed_at":"2026-08-04T01:12:38.857487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:38.939920Z","title":"none identified","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:38.939920Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c6fba75256c5e76b4f3a056e7150c90fb0dd6f0e41cffbdfc7bc7f24d3a2a984","observation_id":"62aa1b8c-60ad-45aa-9b9b-2d5cc21f14e3","resolution":{"observed_at":"2026-08-04T01:12:38.939920Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.018297Z","title":"reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.018297Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:f5c8015ec176d05cd3a7f38fbc407fa8d8a76c7d3636803517d17e882fb09f34","observation_id":"3377049d-f4bc-4c76-8fb7-557fa03b9c9e","resolution":{"observed_at":"2026-08-04T01:12:39.018297Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.092541Z","title":"Order from most to least important","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.092541Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:8e632e896a23361e6d8d23e761c835a91d4a5b0446d9a80d12fbda7d500a0cdc","observation_id":"e2a5d1a1-0e80-419f-9eef-e66ae93408dc","resolution":{"observed_at":"2026-08-04T01:12:39.092541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.153063Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.153063Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:c56ac98abf90908fab87b67216760ac44031fdb5d962499c13642ab06376e7ff","observation_id":"f10004a5-26a9-4d39-95d1-277af5e60038","resolution":{"observed_at":"2026-08-04T01:12:39.153063Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.230523Z","title":"Status:","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.230523Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:cee20270e8afbda9455c7ad8315cb817aa36cec89af6a716646df56789512f7f","observation_id":"90dc9e6e-7d6c-4c87-8624-ff0085958f7c","resolution":{"observed_at":"2026-08-04T01:12:39.230523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.310212Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.310212Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:63fea2452cadbabd8d0ba91e6a30318b0e0228887be8068739dcaf54b58af1bf","observation_id":"53aba8b1-c42c-45e0-9c21-a9a2afb989b4","resolution":{"observed_at":"2026-08-04T01:12:39.310212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.419533Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.419533Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:3757febc74e87902d14260a82c367ab8a3681613db4cb20a2d5e53af50236a42","observation_id":"09cd7c8b-48af-4bba-ade3-477b10cbe395","resolution":{"observed_at":"2026-08-04T01:12:39.419533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.476352Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.476352Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:dc791c82f326d535e2387de0822561d0ff4c0c00643816f5ee85045861dccc38","observation_id":"00a09f38-aa8c-413d-a4b4-396c781e0e26","resolution":{"observed_at":"2026-08-04T01:12:39.476352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-04T01:12:39.559055Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-04T01:12:39.559055Z"},"links":{"citing_paper":"/paper/2608.00155"},"observation_digest":"sha256:a13b28b56faa413c7e3a0319d2a48c84cbff237ac38fbe8a9599eb7658e8c5d9","observation_id":"6984840c-21ed-47f3-a1db-0713b084276d","resolution":{"observed_at":"2026-08-04T01:12:39.559055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.00155","last_updated":"2026-07-31T17:50:08Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T19:09:58.126523Z","submitted_at":"2026-07-31T17:50:08Z","title":"AgentStream: How Well Do Self-Evolving LLM Agents Perform Under Streaming Tasks?"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":99,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":102},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 102 outbound references and 0 inbound Pith citation observations for arXiv:2608.00155."}