{"as_of":"2026-08-08T13:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fa5c4b4949225316a7cbe8c81608b0895b7708a9486863e4dc1330e1f972a2fc","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":29,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:31:27.304146Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2504.19678","last_updated":"2026-03-06T19:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T11:08:22Z","title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","version":2},"reference_index":126,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:37.873567Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2504.19678"},"observation_digest":"sha256:15617302501e8140f529d49cd873fa77d3021f809b1492cc9ec71dbb390df593","observation_id":"e917c6d9-ae85-4aba-8155-8e4637ba99a5","resolution":{"observed_at":"2026-05-15T02:57:38.485174Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-06T18:31:27.304146Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07998","last_updated":"2025-08-27T07:42:56Z","snapshot_observed_at":"2026-08-07T15:21:56.692035Z","submitted_at":"2025-07-10T17:59:55Z","title":"PyVision: Agentic Vision with Dynamic Tooling","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T18:31:27.304146Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2507.07998"},"observation_digest":"sha256:7f93aa923ddabd8811034bdb1bc4bff9ce357d02fc0af29b7a9cc88171ec1906","observation_id":"b4ffdfc9-7646-4a7d-b84c-26b6d3f8d008","resolution":{"observed_at":"2026-08-06T18:31:27.304146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T11:39:10.799080Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02655","last_updated":"2026-06-03T16:39:45Z","snapshot_observed_at":"2026-08-08T03:37:57.969187Z","submitted_at":"2025-09-02T15:13:14Z","title":"BioBlue: Systematic runaway-optimiser-like LLM failure modes on biologically and economically aligned AI safety benchmarks for LLMs with simplified observation format","version":3},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T11:39:10.799080Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2509.02655"},"observation_digest":"sha256:2b61df7ef6a2fa4543d59e5f610136cea444894cae57cf68394b0f18b18b1875","observation_id":"4877a63d-1dff-4ce2-b53b-1d5ed6dc5a8d","resolution":{"observed_at":"2026-08-05T11:39:10.799080Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-04T21:29:55.517898Z","title":"and Elwood, R","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2509.07961","last_updated":"2026-05-23T11:23:56Z","snapshot_observed_at":"2026-08-08T01:52:29.844748Z","submitted_at":"2025-09-09T17:48:44Z","title":"Probing the Preferences of a Language Model: Integrating Verbal and Behavioral Tests of AI Welfare","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T21:29:55.517898Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2509.07961"},"observation_digest":"sha256:9f2348175344bbb443b2bffa7334c37c6d81e4f695c17b16a395b5dabbbfbd88","observation_id":"32327f00-ae59-4aab-aae6-910381a70a9f","resolution":{"observed_at":"2026-08-04T21:29:55.517898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-03T21:20:35.290693Z","title":"and Petersson, L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2511.15830","last_updated":"2026-07-20T14:51:40Z","snapshot_observed_at":"2026-08-07T04:36:26.858564Z","submitted_at":"2025-11-19T19:38:05Z","title":"Mini Amusement Parks (MAPs): A Testbed for Modelling Business Decisions","version":3},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-03T21:20:35.290693Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2511.15830"},"observation_digest":"sha256:df96e67cbae2c04db1e6836b50f00e1dfcd75a5179bd1eba1016ce78bd1f0dfe","observation_id":"c124a8f3-92c1-4ae3-b4e8-b0464d687c5b","resolution":{"observed_at":"2026-08-03T21:20:35.290693Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-03T04:01:46.074199Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.06357","last_updated":"2026-05-24T02:03:19Z","snapshot_observed_at":"2026-08-08T04:54:32.205643Z","submitted_at":"2026-02-06T03:39:15Z","title":"LLM-SAA: LLM-persona Generated Distributions for Decision-making","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-03T04:01:46.074199Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2602.06357"},"observation_digest":"sha256:2cc137cecdf1d61eb07af980e6c3959a409e4256423497aa8083aae96e0b0663","observation_id":"7ee1af49-dc9c-4beb-9424-7cf3aa09b1e8","resolution":{"observed_at":"2026-08-03T04:01:46.074199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2602.09514","last_updated":"2026-05-09T05:05:58Z","snapshot_observed_at":"2026-08-07T18:15:27.312882Z","submitted_at":"2026-02-10T08:12:23Z","title":"EcoGym: Evaluating LLMs for Long-Horizon Plan-and-Execute in Interactive Economies","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T05:53:29.860037Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2602.09514"},"observation_digest":"sha256:839850afae3e5a822fb324103267d4f1ceff8b1e1214e5f01c90b7907aebea19","observation_id":"9329eed0-28e2-4aaf-9c6b-9b73b2a1c756","resolution":{"observed_at":"2026-05-16T05:57:24.662243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2602.15763","last_updated":"2026-02-24T10:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-17T17:50:56Z","title":"GLM-5: from Vibe Coding to Agentic Engineering","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T05:46:40.836161Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2602.15763"},"observation_digest":"sha256:b173154bf3c1aa46fa1dc04f2f844f0db6505c8eb15152746f781e8643a8e237","observation_id":"80edd497-0d36-49de-9428-7ab2ca384f26","resolution":{"observed_at":"2026-05-11T05:46:41.129911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-07-13T15:09:29.436835Z","title":"[Cai et al.(2023)] Tianle Cai, Xuezhi Wang, Tengyu Ma, Xinyun Chen, and Denny Zhou","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.00392","last_updated":"2026-07-04T19:33:13Z","snapshot_observed_at":"2026-07-30T09:24:22.617019Z","submitted_at":"2026-04-01T02:21:55Z","title":"Beyond Task Completion: A Verification-vs.-Conformance Gap in Tool-Evolving Agents","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-13T15:09:29.436835Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.00392"},"observation_digest":"sha256:c4300353acc4dc57f81d62b4622d05d71cef1fce107a58696712ca6b3a46f1f6","observation_id":"b5035e18-9b49-4d02-810c-bb368d148b38","resolution":{"observed_at":"2026-07-13T15:09:29.436835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2604.01444","last_updated":"2026-04-03T15:46:12Z","snapshot_observed_at":"2026-07-06T22:51:35.522923Z","submitted_at":"2026-04-01T22:38:38Z","title":"Cooking Up Risks: Benchmarking and Reducing Food Safety Risks in Large Language Models","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T22:02:12.555644Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.01444"},"observation_digest":"sha256:ddb231b8848630fda620213f04eeb39a4c11c2e91db970fb3ac3a0d68db78177","observation_id":"70d12f98-1c10-4ba0-8fd5-7f5d50dd68f3","resolution":{"observed_at":"2026-05-13T22:03:20.367050Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2604.07733","last_updated":"2026-04-09T02:29:20Z","snapshot_observed_at":"2026-07-06T22:57:00.904627Z","submitted_at":"2026-04-09T02:29:20Z","title":"CivBench: Progress-Based Evaluation for LLMs' Strategic Decision-Making in Civilization V","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T17:59:21.887765Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.07733"},"observation_digest":"sha256:0d1eb16381b478557752bf0b636465fc640d6fd4d8fdcde773ecfa307b381dbb","observation_id":"ed5787e0-d2e7-485d-893e-4613887c86c6","resolution":{"observed_at":"2026-05-11T05:41:01.844569Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2604.09678","last_updated":"2026-04-03T05:11:05Z","snapshot_observed_at":"2026-07-06T22:58:29.952947Z","submitted_at":"2026-04-03T05:11:05Z","title":"NetAgentBench: A State-Centric Benchmark for Evaluating Agentic Network Configuration","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T18:49:15.486503Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.09678"},"observation_digest":"sha256:3ae1a625f6c642e8e0eae6b89d480b6f0ff9aa4449ecaac9c19f0603fe55987f","observation_id":"44113a13-f4f7-4fcf-af13-fdf52409b9ab","resolution":{"observed_at":"2026-05-13T18:53:08.341871Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2604.15267","last_updated":"2026-07-05T03:19:06Z","snapshot_observed_at":"2026-08-03T17:14:09.113583Z","submitted_at":"2026-04-16T17:40:30Z","title":"CoopEval: Benchmarking Cooperation-Sustaining Mechanisms and LLM Agents in Social Dilemmas","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T09:27:02.965559Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.15267"},"observation_digest":"sha256:936cf2e92095c6a79f593d7102aa33288fbfa5b67f071a914c7739badce5cefd","observation_id":"5e7f2a29-2cc6-41d3-9f64-8cf0c5e25773","resolution":{"observed_at":"2026-05-10T09:28:38.701078Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-07-12T19:45:34.692776Z","title":"Technical Report","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2604.15267","last_updated":"2026-07-05T03:19:06Z","snapshot_observed_at":"2026-08-03T17:14:09.113583Z","submitted_at":"2026-04-16T17:40:30Z","title":"CoopEval: Benchmarking Cooperation-Sustaining Mechanisms and LLM Agents in Social Dilemmas","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-12T19:45:34.692776Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.15267"},"observation_digest":"sha256:aa323c8d3659038b3190546b8280b26ce60125e46e3ce5d39ceb9139b9f54c83","observation_id":"094ade2b-8ef5-4766-b959-46212130cab2","resolution":{"observed_at":"2026-07-12T19:45:34.692776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2604.19837","last_updated":"2026-04-21T08:14:09Z","snapshot_observed_at":"2026-08-04T06:15:32.284780Z","submitted_at":"2026-04-21T08:14:09Z","title":"Forage V2: Knowledge Evolution and Transfer in Autonomous Agent Organizations","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-10T03:14:01.097146Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.19837"},"observation_digest":"sha256:24fa1c2210c2c3f75a96ddb580c094454d5922e8fc00e6d67eed9107a9ebba88","observation_id":"fcfb4317-89f8-4629-8821-98b268dfbbcf","resolution":{"observed_at":"2026-05-10T03:14:07.696827Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2604.27043","last_updated":"2026-04-29T17:44:32Z","snapshot_observed_at":"2026-08-03T12:37:08.017930Z","submitted_at":"2026-04-29T17:44:32Z","title":"CL-bench Life: Can Language Models Learn from Real-Life Context?","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-07T09:42:33.635866Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2604.27043"},"observation_digest":"sha256:9fdd7f1257f7bd37ca7124b3625ba61a857625ceeea3f9b8268e56a1effc07ce","observation_id":"7f937a47-e7c0-46c2-9d87-c97643217658","resolution":{"observed_at":"2026-05-12T09:41:27.025477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2605.10310","last_updated":"2026-06-19T14:35:47Z","snapshot_observed_at":"2026-07-06T23:22:18.704329Z","submitted_at":"2026-05-11T10:11:08Z","title":"Positive Alignment: Artificial Intelligence for Human Flourishing","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-15T05:56:56.902705Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2605.10310"},"observation_digest":"sha256:762d83ca10663989b873df5ff2a32245be75951aca8e2683a4175e5b53836e0d","observation_id":"f5b9b71b-94e5-4585-bf4d-7f13bc5405a0","resolution":{"observed_at":"2026-05-15T05:59:47.873226Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2605.17698","last_updated":"2026-05-17T23:36:55Z","snapshot_observed_at":"2026-07-06T23:28:41.111402Z","submitted_at":"2026-05-17T23:36:55Z","title":"Agent Bazaar: Enabling Economic Alignment in Multi-Agent Marketplaces","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T13:25:48.728238Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2605.17698"},"observation_digest":"sha256:545d737eee68f95588dac434244b5a6db6a6f37cf9de21bb302343579ff99654","observation_id":"cf28f79f-6def-4060-99f5-446639afa09f","resolution":{"observed_at":"2026-05-20T13:28:19.054534Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2606.00341","last_updated":"2026-05-29T20:29:35Z","snapshot_observed_at":"2026-08-07T20:25:37.926234Z","submitted_at":"2026-05-29T20:29:35Z","title":"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T23:03:36.851403Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.00341"},"observation_digest":"sha256:7ce8bfc16d1675d9fd9499712137610a2790a5b08d5dfde05e99599960f31808","observation_id":"5ad775ec-7465-4ab4-8e7b-85ec66537527","resolution":{"observed_at":"2026-07-01T19:16:00.246835Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2606.18543","last_updated":"2026-07-22T06:45:58Z","snapshot_observed_at":"2026-08-08T11:07:46.747541Z","submitted_at":"2026-06-16T23:37:52Z","title":"CEO-Bench: Can Agents Play the Long Game?","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-06-27T00:19:47.396643Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.18543"},"observation_digest":"sha256:8557320428fc5428b7d27fef3761b101add36cd14d77d8b80a1bdec07bd0d56a","observation_id":"33b4960e-f1b0-44ee-9365-1bb344d6df77","resolution":{"observed_at":"2026-07-03T21:38:58.693981Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-02T11:01:16.111772Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.18543","last_updated":"2026-07-22T06:45:58Z","snapshot_observed_at":"2026-08-08T11:07:46.747541Z","submitted_at":"2026-06-16T23:37:52Z","title":"CEO-Bench: Can Agents Play the Long Game?","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-02T11:01:16.111772Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.18543"},"observation_digest":"sha256:3cb73def3366a4833c1d695917b1753f6be861fa6874c83b6c57f149c3b1d5bd","observation_id":"ee76460d-f7df-49d9-a7a4-fdc5bcfc724e","resolution":{"observed_at":"2026-08-02T11:01:16.111772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2606.19308","last_updated":"2026-06-17T17:31:06Z","snapshot_observed_at":"2026-08-04T23:52:08.389322Z","submitted_at":"2026-06-17T17:31:06Z","title":"Enhancing Decision-Making with Large Language Models through Multi-Agent Fictitious Play","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-26T20:57:49.840546Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.19308"},"observation_digest":"sha256:648a66d20b46ed59b584ec7f12b46f8287e93f29ef6e49bfe472512d9db21922","observation_id":"a76bf20e-fa30-4e2f-83b3-ea5bd853df8e","resolution":{"observed_at":"2026-07-04T00:49:18.490566Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2606.19613","last_updated":"2026-06-17T21:36:09Z","snapshot_observed_at":"2026-08-07T13:16:30.354063Z","submitted_at":"2026-06-17T21:36:09Z","title":"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-26T19:47:59.090249Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.19613"},"observation_digest":"sha256:8cd107bd04b1078fe817aa6a0e6f3fea48563bcf8dc4e5372bf6415d95faba23","observation_id":"0746d08a-99fe-4367-bc8d-99bb79d794f7","resolution":{"observed_at":"2026-07-04T02:19:23.915566Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2606.22909","last_updated":"2026-06-22T06:49:38Z","snapshot_observed_at":"2026-07-06T23:57:42.632111Z","submitted_at":"2026-06-22T06:49:38Z","title":"Graph-Enhanced Large Language Models for Spatial Search","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-26T06:46:29.757188Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.22909"},"observation_digest":"sha256:6e12ef4b273228bfaeecafe1511881327e16a7ab5bb14e4dfb35fb6b7d1331b5","observation_id":"51fbd126-8dbb-4a01-ab59-49c10fb4055d","resolution":{"observed_at":"2026-07-04T12:29:51.950322Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":"2502.15840","doi":"10.48550/arxiv.2502.15840","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":"ArXiv.org","work_id":"6ae40d2d-f1a8-4d13-bf9e-a0a98fc81964","year":2025},"citing_paper":{"arxiv_id":"2606.31916","last_updated":"2026-06-30T16:22:12Z","snapshot_observed_at":"2026-08-02T10:24:05.379334Z","submitted_at":"2026-06-30T16:22:12Z","title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-07-01T05:40:54.002702Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2606.31916"},"observation_digest":"sha256:3a5d4c68441d8272ca1cacc8f1becf31e60e4c79d6e20dfd15bccd776e190547","observation_id":"3526cecc-8a5a-425d-b370-31f8a8788bdf","resolution":{"observed_at":"2026-07-01T10:15:44.678930Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-07-15T07:32:36.632338Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.12125","last_updated":"2026-07-13T20:17:43Z","snapshot_observed_at":"2026-08-07T17:14:55.707512Z","submitted_at":"2026-07-13T20:17:43Z","title":"Faster AI, Uneven Frontier: Rapid Crossings, a Jagged Frontier, and the Repositioning of Human Judgment","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-15T07:32:36.632338Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2607.12125"},"observation_digest":"sha256:4843c90972cc2ae2be8667be1e4bd5fabfbd565369553e22ce4246104b4cb3b9","observation_id":"d05376fd-dcb6-4499-8295-d0052e97d8be","resolution":{"observed_at":"2026-07-15T07:32:36.632338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-02T04:42:05.920354Z","title":"Backlund and L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.13618","last_updated":"2026-07-15T09:08:27Z","snapshot_observed_at":"2026-08-02T14:36:13.285477Z","submitted_at":"2026-07-15T09:08:27Z","title":"STOCKTAKE: Measuring the Gap Between Perception and Action in LLM Agents with a Fair Oracle","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T04:42:05.920354Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2607.13618"},"observation_digest":"sha256:924d4e879c4804e90fca74d1e33f2a77e7d9113f7db793b59f8e417d65967a89","observation_id":"1e3dff5a-2ea5-48cd-8372-785d5005be33","resolution":{"observed_at":"2026-08-02T04:42:05.920354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-02T00:36:15.686510Z","title":"Vending-bench: A benchmark for long-term coherence of autonomous agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.14989","last_updated":"2026-07-16T13:38:07Z","snapshot_observed_at":"2026-08-06T20:42:21.683853Z","submitted_at":"2026-07-16T13:38:07Z","title":"OmniaBench: Benchmarking General AI Agents Across Diverse Scenarios","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T00:36:15.686510Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2607.14989"},"observation_digest":"sha256:1842b897c32c76489e7b4a8e00fee7403beb2695fc2248af17a35f1750f51528","observation_id":"f92f1d4b-46d7-4c3f-81e9-094a084376a9","resolution":{"observed_at":"2026-08-02T00:36:15.686510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.15840","snapshot_observed_at":"2026-08-01T02:36:07.191363Z","title":"Backlund and L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25398","last_updated":"2026-08-03T20:36:31Z","snapshot_observed_at":"2026-08-07T23:09:37.718671Z","submitted_at":"2026-07-28T07:58:07Z","title":"HANDBOOK.md: A Benchmark for Long-Context Agentic Instruction Following","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T02:36:07.191363Z"},"links":{"cited_paper":"/paper/2502.15840","citing_paper":"/paper/2607.25398"},"observation_digest":"sha256:e761de4eaeaceb6902cc77df4ea90dfcc92ec76c887fb0d072f7570e147cc498","observation_id":"854d0ebe-bbf0-4d65-87de-47cbcbb6674e","resolution":{"observed_at":"2026-08-01T02:36:07.191363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.15840/citation-record","integrity":"/paper/2502.15840/integrity","json":"/paper/2502.15840/citation-record.json","paper":"/paper/2502.15840"},"outbound":[],"paper":{"arxiv_id":"2502.15840","last_updated":"2025-02-20T15:52:29Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-07T18:01:38.589431Z","submitted_at":"2025-02-20T15:52:29Z","title":"Vending-Bench: A Benchmark for Long-Term Coherence of Autonomous Agents"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 29 inbound Pith citation observations for arXiv:2502.15840."}