{"as_of":"2026-08-07T05:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d80b275efb9351ac7def90b62f84e6c2f16731afb8e49d5b3b49c893e5c14fee","coverage":[{"denominator":101,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-20T20:19:21.824216Z","state":"measured"},{"denominator":101,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":101,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:45:35.202876Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-07T04:45:37.615549Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"cited_work":{"arxiv_id":"2605.14133","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.14133","snapshot_observed_at":"2026-08-07T04:45:37.615549Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","venue":"cs.AI","work_id":"c49596b3-75e6-4f92-9528-b4904e151f4e","year":2026},"citing_paper":{"arxiv_id":"2608.06352","last_updated":"2026-08-06T17:53:18Z","snapshot_observed_at":"2026-08-07T05:30:58.975159Z","submitted_at":"2026-08-06T17:53:18Z","title":"CalibForge: Adversarial Solver Calibration for Scaling Learnable Terminal Tasks","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T04:45:35.202876Z"},"links":{"cited_paper":"/paper/2605.14133","citing_paper":"/paper/2608.06352"},"observation_digest":"sha256:1c4daf9a3f3b08489c80e05d696b3032728f86fabb3b6a268aa4991e8252c481","observation_id":"07c4cc93-02ca-4dce-b1df-3ebeccb6d82f","resolution":{"observed_at":"2026-08-07T04:45:37.620599Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.14133/citation-record","integrity":"/paper/2605.14133/integrity","json":"/paper/2605.14133/citation-record.json","paper":"/paper/2605.14133"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.12110","last_updated":"2025-10-08T01:46:37Z","snapshot_observed_at":"2026-08-03T02:27:06.991396Z","submitted_at":"2025-02-17T18:36:14Z","title":"A-MEM: Agentic Memory for LLM Agents","version":11},"cited_work":{"arxiv_id":"2502.12110","doi":"10.48550/arxiv.2502.12110","metadata_source":"pith","pith_arxiv_id":"2502.12110","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A-MEM: Agentic Memory for LLM Agents","venue":"cs.CL","work_id":"3b98feb2-fdb1-479a-bbe4-2c298a4592e2","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2502.12110","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:00d64a42a025a9b3e856cef187a240cac6f88538713f6f90d17aeb167a516481","observation_id":"c506d781-5242-4333-ad29-2a91399bc01a","resolution":{"observed_at":"2026-05-20T20:23:43.232826Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.13956","last_updated":"2025-01-20T16:52:48Z","snapshot_observed_at":"2026-07-29T15:53:45.545304Z","submitted_at":"2025-01-20T16:52:48Z","title":"Zep: A Temporal Knowledge Graph Architecture for Agent Memory","version":1},"cited_work":{"arxiv_id":"2501.13956","doi":"10.48550/arxiv.2501.13956","metadata_source":"pith","pith_arxiv_id":"2501.13956","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zep: A Temporal Knowledge Graph Architecture for Agent Memory","venue":"cs.CL","work_id":"515c933e-12ae-439d-a7ff-c07fee482dfb","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2501.13956","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:094291c2cc227273dada59a87591d81a603b579c72f3115a383ae4160217b253","observation_id":"9d552512-599c-43ed-b6fa-b14fe3918dea","resolution":{"observed_at":"2026-05-20T20:23:43.266224Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T11:57:04.111004Z","title":"Proceedings of the AAAI conference on artificial intelligence , volume=","venue":null,"work_id":"a60807ef-dd92-4b41-94d6-cd81a67ea2ec","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:e56c7a976faaf23756efbeec6e50800198c9a329c3a0f49efbea743334016b2b","observation_id":"6dadaf98-a18c-41cb-9423-f9c534de58ca","resolution":{"observed_at":"2026-05-20T20:23:43.617096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:52:55.746423Z","title":"Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing , pages=","venue":null,"work_id":"03eb5914-f16b-41a5-a3eb-1ec8872212cc","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:4a9b83fa721989842bd164c2da1f536001bc9940d2b71889baab76c0ec8b9daf","observation_id":"b852cc17-53c5-45ea-b847-94f550b98c3a","resolution":{"observed_at":"2026-05-20T20:23:43.615410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T11:57:04.122514Z","title":"author=","venue":null,"work_id":"28a11e01-274f-4df2-ba22-74b5dc701483","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:6a2228d1636f2b61b39cc9b7b1e700d367322625f77b1d09de4b062380b0d432","observation_id":"ffbfee9d-f7f3-4273-8c74-1ce8b8c3bbca","resolution":{"observed_at":"2026-05-20T20:23:43.619234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18543","last_updated":"2026-06-10T02:43:26Z","snapshot_observed_at":"2026-08-06T19:53:43.422577Z","submitted_at":"2026-04-20T17:36:49Z","title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","version":4},"cited_work":{"arxiv_id":"2604.18543","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.18543","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ClawEnvKit: Automatic Environment Generation for Claw-Like Agents","venue":"cs.AI","work_id":"bbc513e6-c637-4529-a8ea-bd6d6a71e9f9","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2604.18543","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:4a1dff6f8b0d39de5fefa23d0102b0652b3d8d7b77142e694da94954a430f3c5","observation_id":"98b5c3c9-90de-4059-9c5a-926d7ecce04c","resolution":{"observed_at":"2026-05-20T20:23:43.268782Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.06132","last_updated":"2026-05-07T14:06:23Z","snapshot_observed_at":"2026-07-06T22:54:40.349479Z","submitted_at":"2026-04-07T17:43:18Z","title":"Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents","version":3},"cited_work":{"arxiv_id":"2604.06132","doi":"10.48550/arxiv.2604.06132","metadata_source":"pith","pith_arxiv_id":"2604.06132","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents","venue":"cs.AI","work_id":"57acc3ec-f4c3-49ab-bd0f-5aab91002df9","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2604.06132","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:e31a6dfb6641e6070ac5bed692b019d1566764f97ade13eba01b802c82ee83fb","observation_id":"0a46bae0-9e2d-4c6e-9985-dcac2444c112","resolution":{"observed_at":"2026-05-20T20:23:43.205856Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:d4255cc97cd04bcca8487421c0743bf4b99924abc9e419865eabd18392ff1304","observation_id":"6ea8b4e0-dd3e-471f-9a61-85c3ec0be6e3","resolution":{"observed_at":"2026-05-20T20:23:43.298529Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"080c7d97-d64f-493d-b9a5-8539570bc321","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:e9fe3733f65a2862cf77229bc521b04b1b3d778897773e57a44b9c2da62863fd","observation_id":"bca6a4ab-ed63-4795-8e11-88a06f7c86b8","resolution":{"observed_at":"2026-05-20T20:23:43.571077Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.05172","last_updated":"2026-04-08T09:27:21Z","snapshot_observed_at":"2026-07-06T22:54:00.211106Z","submitted_at":"2026-04-06T21:09:06Z","title":"ClawsBench: Evaluating Capability and Safety of LLM Productivity Agents in Simulated Workspaces","version":2},"cited_work":{"arxiv_id":"2604.05172","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.05172","snapshot_observed_at":"2026-07-03T06:07:41.480446Z","title":"ClawsBench: Evaluating Capability and Safety of LLM Productivity Agents in Simulated Workspaces","venue":"cs.AI","work_id":"1d0d802c-947a-4c92-8995-c392b01261d7","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2604.05172","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:269f2b8ff89a1a844efb181e930593b9eb9647972112cf3c25c4b3de71f5f09e","observation_id":"22bdc956-75a4-4be0-96a9-6eba6d84e67b","resolution":{"observed_at":"2026-05-20T20:23:43.305322Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.04202","last_updated":"2026-05-16T15:44:46Z","snapshot_observed_at":"2026-08-01T03:44:06.431113Z","submitted_at":"2026-04-05T17:55:23Z","title":"ClawArena: Benchmarking AI Agents in Evolving Information Environments","version":2},"cited_work":{"arxiv_id":"2604.04202","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.04202","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ClawArena: Benchmarking AI Agents in Evolving Information Environments","venue":"cs.LG","work_id":"177dafeb-01c2-4b16-96f2-d33a422009b9","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2604.04202","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:5a1c693a54e66fa769b204713f029060a1371d47ed92308966aef04f63e2dcfc","observation_id":"52cb93c9-9be5-468c-9956-a1a6de76fe6d","resolution":{"observed_at":"2026-05-20T20:23:43.252073Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T17:56:26.106824Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"bbd406a3-5c71-400c-8be6-a6512f5ba309","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:8ee7cb6d976e121d8be645b7ad3684972111297c29027fb30cbe6ff092ca7533","observation_id":"2ab436f9-2e60-425c-8d33-3fc0ae587a5b","resolution":{"observed_at":"2026-05-20T20:23:43.600675Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T15:16:18.969978Z","title":"ACM transactions on intelligent systems and technology , volume=","venue":null,"work_id":"49d9503c-8827-478f-880d-2b4b98b23300","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:f9fbbbfb124c2e651061a1370fa46265fe5a3dcf90f8cb2c51ae409795834058","observation_id":"b9857001-9f9c-4855-a558-7e3b46a98960","resolution":{"observed_at":"2026-05-20T20:23:43.611326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-07-06T16:00:46.542753Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":"2307.16789","doi":"10.48550/arxiv.2307.16789","metadata_source":"pith","pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","venue":"cs.AI","work_id":"3c555b48-a4d9-42dd-9fdd-0f6018fbe9cb","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:bb1f25c80c5c6967b4a39ace4d0aa390d25366b3521a54a1c825771e6c7facef","observation_id":"d0f7884a-c7bf-4bb4-bc59-92c32473ec89","resolution":{"observed_at":"2026-05-20T20:23:43.257814Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:33.730697+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:33.730697+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T14:07:07.078437Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"9e679e73-1ab3-49cd-a880-6583acefe275","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:ae142e8891e6b662453c25ec7864ea06f994914ea4ff15f37558eff7ea0a2463","observation_id":"19f9bdda-d2e2-4ca3-a8ff-cd8a95522b96","resolution":{"observed_at":"2026-05-20T20:23:43.588923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T02:21:49.356376Z","title":"The Twelfth International Conference on Learning Representations , year=","venue":null,"work_id":"5de70941-c7c2-4d29-9679-775f7dd66603","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:9ff8fe05c4f5100265453607ebb71f4847617fa0444c29fc11b1b19c9282f078","observation_id":"599ed05a-99de-4a04-b221-93ae81f370f7","resolution":{"observed_at":"2026-05-20T20:23:43.592691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T13:36:19.987327Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"d83a3b30-39ae-4c08-a2ec-1e6eefc202b3","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:2a1078dba357f511a164e6499dcf4381de1472a4022e7c4c26312942a8ccf92a","observation_id":"57098780-5d3c-4871-b7d9-281a9bef4d20","resolution":{"observed_at":"2026-05-20T20:23:43.584867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-06T20:36:41.418114Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":"2308.03688","doi":"10.1109/fllm63129.2024.10852426","metadata_source":"pith","pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AgentBench: Evaluating LLMs as Agents","venue":"cs.AI","work_id":"a37549b4-4c94-412d-acc4-4efeb08509be","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:995566a0c8a8539d6ba821c6e35fe0d441f0cbc690276bd7a4c984d3645b188c","observation_id":"b1e0d0b2-4ee6-4a1d-a586-c5fcec4ba5d2","resolution":{"observed_at":"2026-05-20T20:23:43.285765Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":"2310.06770","doi":"10.1145/512927.512945","metadata_source":"pith","pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","venue":"cs.CL","work_id":"d0effe15-a689-441a-8e3f-ea35f1c4e4b1","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:8e8ef060d63b7d07d2cc3647138c4abc64ddc68704e71e7524ed244489d8c603","observation_id":"849e6e08-40dd-4f34-b595-e7dbbb06b952","resolution":{"observed_at":"2026-05-20T20:23:43.293321Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-06T12:47:01.809185Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":"2307.13854","doi":"10.48550/arxiv.2307.13854","metadata_source":"pith","pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","venue":"cs.AI","work_id":"7058ffd2-a339-4102-89eb-248eeb074652","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:ea8a0a78331a7a83430c490b577fa3c8f96287a9d14c933808012823884e6afb","observation_id":"4875af07-62ff-4316-b2bc-290c472ab716","resolution":{"observed_at":"2026-05-20T20:23:43.260600Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:18.85917+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:18.85917+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T03:08:05.536825Z","title":"Proceedings of the National Academy of Sciences , volume=","venue":null,"work_id":"3c44ab01-4542-4d51-8e01-6d550efb31da","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:9dcacb7515361feb7c1d6d211c60ee0ef1f3d332315d7b06c80561dc26ef2c7e","observation_id":"680341bd-9a72-488d-ba5d-48c85062a6f3","resolution":{"observed_at":"2026-05-20T20:23:43.582909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2026 , organization =","venue":null,"work_id":"1d5a03b0-6a40-4e67-b396-b2d7c1309521","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:52899289e8bbafb4621866083723c544a45925df6270e2cdec0a23e1c52465c0","observation_id":"e66d6e89-3311-4752-92ee-d61e00d58e49","resolution":{"observed_at":"2026-05-20T20:23:43.587127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , year=","venue":null,"work_id":"d8363f4b-f82d-46b0-9f5f-970bcd5e25ab","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:588d97d6943b78194b7c8170a85d0342a605ea6ab04a5bb1aa4cf9a1809b8bc7","observation_id":"118a9a4d-f76f-411d-8306-82a721ce0e51","resolution":{"observed_at":"2026-05-20T20:23:43.590580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient Lifelong Learning with","venue":null,"work_id":"f73d4359-13fa-4806-9f16-e35d114751b1","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:790b64edb47908746fa26d076b463eeca749c2739e97451c3e475acc8a567b4d","observation_id":"b6726224-d691-4041-a418-b413cfdf3889","resolution":{"observed_at":"2026-05-20T20:23:43.607726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning , year=","venue":null,"work_id":"e1271d2f-a922-45e3-991e-298b0caf7bf6","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:0794d8b5773377b4f2d8f3ef1d523c33f68ec598ea20bad291090b1e85f47479","observation_id":"72d81367-6f0d-404e-ab86-e04a6be6f732","resolution":{"observed_at":"2026-05-20T20:23:43.628374Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1611.02779","last_updated":"2016-11-10T01:17:36Z","snapshot_observed_at":"2026-08-02T07:58:55.035919Z","submitted_at":"2016-11-09T00:13:29Z","title":"RL$^2$: Fast Reinforcement Learning via Slow Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"1611.02779","doi":"10.48550/arxiv.1611.02779","metadata_source":"pith","pith_arxiv_id":"1611.02779","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RL$^2$: Fast Reinforcement Learning via Slow Reinforcement Learning","venue":"cs.AI","work_id":"f9b75698-414f-456e-a2e6-dbe568b2693d","year":2016},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/1611.02779","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:079196224de75f853023d7890096f58b177d0324a70aa110fbb2ba40d582781f","observation_id":"12348fa6-adf2-4fb6-9276-cc8b6c8f9c71","resolution":{"observed_at":"2026-05-20T20:23:43.219405Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Machine Learning , year=","venue":null,"work_id":"1925e58d-601e-4ed3-bf3d-7f05006394d0","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:2dd359969e7b0707e4be729a9a2a5d97f34ac13b0013e2303ed78bb1652a69a5","observation_id":"80cac00d-ba5f-4267-94ff-f122e2b12672","resolution":{"observed_at":"2026-05-20T20:23:43.574444Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Conference on Learning Representations , year=","venue":null,"work_id":"03f7c26e-7e05-4b26-b974-5b5f9119c428","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:2e0a25a26fb1645110d46fb8be6add4adca3b42169ce61b557a8a8052125f3f0","observation_id":"daa4816c-5381-4dea-b699-d67064d6fac7","resolution":{"observed_at":"2026-05-20T20:23:43.572776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , year=","venue":null,"work_id":"ab9aa563-fac6-4838-bd0d-25786409d9bd","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:c763221226ebe7cc1b2dc13113bbecedf7a1110a8219f108696053068ada9486","observation_id":"72cdc3e9-4b3e-4afe-ab5a-d1242d278fdc","resolution":{"observed_at":"2026-05-20T20:23:43.576140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"460a22fe-86a7-4a8d-9f8d-e9e546e52a15","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:88faf3b99cabfcc56276bfe0c3ec45682ebd51a06f6f93d632ca9b8818d1541d","observation_id":"99b41d65-ae1b-46bd-8666-2c9832c853f0","resolution":{"observed_at":"2026-05-20T20:23:43.577919Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":"2602.02276","doi":"10.48550/arxiv.2602.02276","metadata_source":"pith","pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2.5: Visual Agentic Intelligence","venue":"cs.CL","work_id":"d690be8f-5d53-49b0-b1e7-79668eb8fcdb","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:78eb0dfab82ecfe4a3e30eaaf16c7756f5873446780d41d837508b4eca415c09","observation_id":"1a422a7f-9bca-4147-bd62-267212708230","resolution":{"observed_at":"2026-05-20T20:23:43.288369Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T07:40:48.509479Z","title":"Findings of the Association for Computational Linguistics: ACL 2025 , pages=","venue":null,"work_id":"712f05c9-b732-4708-95f1-0fd1940ee193","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:36d72659f2c878d82f153ee06660589c647da725f5c81275184aeda1db611341","observation_id":"e470b44a-a27a-4dd8-a77e-b05ed3ae569e","resolution":{"observed_at":"2026-05-20T20:23:43.569366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:02:58.716193Z","title":"The eleventh international conference on learning representations , year=","venue":null,"work_id":"d0870e7f-a33a-4020-8109-459b41853021","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:03cd5b0017cf4713ce31372060f2c1c7373c81407b84a65151ca3a7465a2ebe0","observation_id":"93e6a1d2-d111-47bc-bbeb-7937f2d5bec5","resolution":{"observed_at":"2026-05-20T20:23:43.579687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.18746","last_updated":"2025-12-21T14:26:14Z","snapshot_observed_at":"2026-07-06T22:39:43.359149Z","submitted_at":"2025-12-21T14:26:14Z","title":"MemEvolve: Meta-Evolution of Agent Memory Systems","version":1},"cited_work":{"arxiv_id":"2512.18746","doi":"10.48550/arxiv.2512.18746","metadata_source":"pith","pith_arxiv_id":"2512.18746","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MemEvolve: Meta-Evolution of Agent Memory Systems","venue":"cs.CL","work_id":"69fd3342-cd0c-43d2-b12c-2d266dc000b2","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2512.18746","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:c34176553c48b67496447c7609723a82655f5cb83efc8990642ce7069d1dfbcb","observation_id":"60aa9139-216d-4e74-993b-2c095952c349","resolution":{"observed_at":"2026-05-20T20:23:43.224817Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03192","last_updated":"2026-02-12T05:43:57Z","snapshot_observed_at":"2026-08-02T21:01:18.325722Z","submitted_at":"2026-01-06T17:14:50Z","title":"MemRL: Self-Evolving Agents via Runtime Reinforcement Learning on Episodic Memory","version":2},"cited_work":{"arxiv_id":"2601.03192","doi":"10.48550/arxiv.2601.03192","metadata_source":"pith","pith_arxiv_id":"2601.03192","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MemRL: Self-Evolving Agents via Runtime Reinforcement Learning on Episodic Memory","venue":"cs.CL","work_id":"0ab11b49-e934-49e1-9eaf-a25fb7343010","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2601.03192","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:3d8a38285000ac9375c17ac22039e1b3e06e7e711f76a4897dfbdf520714c7c7","observation_id":"030660b5-612f-4ee4-96be-096bd092e41c","resolution":{"observed_at":"2026-05-20T20:23:43.199913Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.25911","last_updated":"2025-09-30T08:02:34Z","snapshot_observed_at":"2026-08-06T02:56:06.576393Z","submitted_at":"2025-09-30T08:02:34Z","title":"Mem-{\\alpha}: Learning Memory Construction via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2509.25911","doi":"10.48550/arxiv.2509.25911","metadata_source":"pith","pith_arxiv_id":"2509.25911","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mem-{\\alpha}: Learning Memory Construction via Reinforcement Learning","venue":"cs.CL","work_id":"4643330b-80ff-49a5-bbb1-e618edaffe35","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2509.25911","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:52930389eb2cf5aacd88bb21c5e2db9c7f2efd36c01d78a0a6ae5027a226beba","observation_id":"5a331ca2-640f-4a46-b4d0-4a1f09a4559d","resolution":{"observed_at":"2026-05-20T20:23:43.202908Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-17T20:50:53.619604+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-17T20:50:53.619604+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2106.09685","last_updated":"2021-10-16T18:40:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-06-17T17:37:18Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","version":2},"cited_work":{"arxiv_id":"2106.09685","doi":"10.4088/pcc.v03n0609","metadata_source":"pith","pith_arxiv_id":"2106.09685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LoRA: Low-Rank Adaptation of Large Language Models","venue":"cs.CL","work_id":"0426219a-789e-4964-adc8-a04538510818","year":2021},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2106.09685","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:2f228286d46eb5f6c7019ae02011ff1af7444d3e1f36ceb339be284d8504b0b7","observation_id":"c391b1c4-9c86-48b9-9792-9e83ce110b92","resolution":{"observed_at":"2026-05-20T20:23:43.275938Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19849","last_updated":"2025-07-26T07:53:11Z","snapshot_observed_at":"2026-07-06T22:03:15.296567Z","submitted_at":"2025-07-26T07:53:11Z","title":"Agentic Reinforced Policy Optimization","version":1},"cited_work":{"arxiv_id":"2507.19849","doi":"10.48550/arxiv.2507.19849","metadata_source":"pith","pith_arxiv_id":"2507.19849","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agentic Reinforced Policy Optimization","venue":"cs.LG","work_id":"6cb0d241-5e4d-44ed-9476-659819ec0681","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2507.19849","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:f8cf6d87cab69d752785ecb2b7fa0191a059376f59945e89eb7b033be7c62299","observation_id":"7ad20e6a-54a8-4071-800a-b47c1866e2e0","resolution":{"observed_at":"2026-05-20T20:23:43.281072Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.14460","last_updated":"2026-05-30T06:08:55Z","snapshot_observed_at":"2026-08-03T21:38:54.007115Z","submitted_at":"2025-11-18T13:03:15Z","title":"Agent-R1: A Unified and Modular Framework for Agentic Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2511.14460","doi":null,"metadata_source":"pith","pith_arxiv_id":"2511.14460","snapshot_observed_at":"2026-07-05T12:30:59.979641Z","title":"arXiv preprint arXiv:2511.14460 , year=","venue":"cs.CL","work_id":"69b25d43-666f-445e-8663-d96d8b033fbb","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2511.14460","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:42fbb15cf32dba913ed8a33f57d232917b2aa1eac269e3187c40f68d4d6de513","observation_id":"8205dc0c-86cf-4630-a061-98a061310350","resolution":{"observed_at":"2026-06-02T02:03:36.165646Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"d089e5ff-b2e6-4903-bca9-4bc90faaaf55","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:64a66dee02bf0aff50c490963deb81773305429760094fc9b2fffebd49f5d778","observation_id":"eba12bb7-94ab-4eb3-a90d-d3a23ea2ba79","resolution":{"observed_at":"2026-05-20T20:23:43.622899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.07671","last_updated":"2019-01-28T19:11:23Z","snapshot_observed_at":"2026-08-04T11:14:14.182296Z","submitted_at":"2018-12-18T22:27:31Z","title":"Deep Online Learning via Meta-Learning: Continual Adaptation for Model-Based RL","version":2},"cited_work":{"arxiv_id":"1812.07671","doi":null,"metadata_source":"pith","pith_arxiv_id":"1812.07671","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep Online Learning via Meta-Learning: Continual Adaptation for Model-Based RL","venue":"cs.LG","work_id":"f026190e-cc2f-4328-aa33-57599caaad3f","year":2018},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/1812.07671","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:ed98d148e52b2d0a929127068f0c3d8276bff2919aa7e651a8e652bc4305b056","observation_id":"7875fc84-9951-4897-afe2-84d764bf24c7","resolution":{"observed_at":"2026-05-20T20:23:43.295748Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International conference on machine learning , pages=","venue":null,"work_id":"80508bc7-7a3a-4529-9fa5-c7dd58267b2a","year":2019},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:10242ece279b6eb6c6d6116b7b288bde5619659bc54a2d424915fe4aff732440","observation_id":"878f693c-75aa-41ca-8276-28e6bff1591b","resolution":{"observed_at":"2026-05-20T20:23:43.647243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T17:12:45.448978Z","title":"Proceedings of the 2024 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":"24d29789-1d88-4b93-ab5d-7e82c1d5df9c","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:6db9df966ec35ad0abec163e761aaaa2e52fbde2ec0439a803f3e2eb9cd046ad","observation_id":"28066ca5-8486-4685-822c-4ca9c2fadb81","resolution":{"observed_at":"2026-05-20T20:23:43.565462Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.07429","last_updated":"2024-09-11T17:21:00Z","snapshot_observed_at":"2026-07-30T13:07:16.440141Z","submitted_at":"2024-09-11T17:21:00Z","title":"Agent Workflow Memory","version":1},"cited_work":{"arxiv_id":"2409.07429","doi":"10.48550/arxiv.2409.07429","metadata_source":"pith","pith_arxiv_id":"2409.07429","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agent Workflow Memory","venue":"cs.CL","work_id":"c81e2e3e-49d4-46f5-9a82-76e8d00fef1d","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2409.07429","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:645d80ad3f6d5837e9b89a9309ec161c33f7ea3ada151a9031ee8587f7f9e589","observation_id":"15682fd4-f984-49ca-afbe-b41b76c83dfd","resolution":{"observed_at":"2026-05-20T20:23:43.303034Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.022945+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.022945+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07398","last_updated":"2025-06-16T08:45:10Z","snapshot_observed_at":"2026-08-07T05:32:29.009208Z","submitted_at":"2025-06-09T03:43:46Z","title":"G-Memory: Tracing Hierarchical Memory for Multi-Agent Systems","version":2},"cited_work":{"arxiv_id":"2506.07398","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.07398","snapshot_observed_at":"2026-07-10T01:36:44.001440Z","title":"G- memory: Tracing hierarchical memory for multi-agent systems","venue":"cs.MA","work_id":"e5c4ec9c-3e43-42d1-a3d7-0deef9c44093","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2506.07398","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:ad3fe71ca905aeaf2b86d694a0056232285d922d71646770037389da941a30ac","observation_id":"e812a28a-df11-4ad8-84db-a08f174907fa","resolution":{"observed_at":"2026-05-20T20:23:43.263414Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T17:56:26.096579Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":"c25e8154-fab2-455c-8a26-56e40aed5d2b","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:6438d93a8b01ec5e0248eb1e73ef8a1a9e4764f6381fcb7a674fe155d1b1f97b","observation_id":"5f66ffa7-ff6a-4a72-a512-7a056785a20f","resolution":{"observed_at":"2026-05-20T20:23:43.567230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"cited_work":{"arxiv_id":"2507.21046","doi":"10.48550/arxiv.2507.21046","metadata_source":"pith","pith_arxiv_id":"2507.21046","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","venue":"cs.AI","work_id":"f5de9511-98bf-411c-9eba-e8b7a914ec18","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2507.21046","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:c0426494edcd3f8adbfbd7f0e5a08fb2b87c7bf756f5dfe72fca7ddd7db7ef80","observation_id":"2eaeec4f-a6e7-4d62-addc-9ef1969c9cce","resolution":{"observed_at":"2026-05-20T20:23:43.213760Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.122082+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.122082+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.16043","doi":"10.48550/arxiv.2511.16043","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agent0: Unleashing self-evolving agents from zero data via tool-integrated reasoning","venue":"ArXiv.org","work_id":"d7e7b19a-a84a-42be-ad4a-550238c4747c","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:89d7213ebe47e23c7843a69f3f711847762c14d48fe6229cb29d2102963cd2cd","observation_id":"45d3bdba-a0fa-48f3-a05a-49fa57729609","resolution":{"observed_at":"2026-05-20T20:23:43.194138Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.19900","doi":"10.48550/arxiv.2511.19900","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agent0 -vl: Exploring self -evolving agent for tool -integrated vision -language reasoning","venue":"ArXiv.org","work_id":"a0b54170-89b2-4a4b-b390-b92a36cb244b","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:d6c9d32be9f7713705c8aaa6466ba1fefb42186ee72b87a33a81734ba9842fbe","observation_id":"d14f72ef-54cb-4916-8734-16a0957f4d5a","resolution":{"observed_at":"2026-05-20T20:23:43.254983Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-30T14:25:28.966971+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-30T14:25:28.966971+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Neural networks , volume=","venue":null,"work_id":"f27df2dc-19de-487d-9c90-fc38e1fe5cf3","year":2019},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:a15d5f973da8af2e92182ae956a075fc3ac44695bede1e2d621634e5eb1aae5a","observation_id":"2de9dac4-8635-48b2-8ffd-f1451a045931","resolution":{"observed_at":"2026-05-20T20:23:43.645568Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T07:30:47.231113Z","title":"Proceedings of the IEEE/CVF conference on computer vision and pattern recognition , pages=","venue":null,"work_id":"66233412-a909-48f8-9987-ed2262e4ce4a","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:0a0a06608a75daf973c49b0876ab8c9d4c48b8fccb307f896c5d39f2130ac9f9","observation_id":"c8f9b9b9-79e2-4aaf-ac5f-112e18ef8d2d","resolution":{"observed_at":"2026-05-20T20:23:43.649194Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T17:32:42.734909Z","title":"IEEE transactions on pattern analysis and machine intelligence , volume=","venue":null,"work_id":"54da80ee-6cdc-44dc-8258-568520fa03ee","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:dcb5ade90baba7086a7e7a53aa5f20e4187af356b3bfe285fbf9a0470cd4034c","observation_id":"64b4efdd-ebd0-495f-acb6-e8cdf66fcbfe","resolution":{"observed_at":"2026-05-20T20:23:43.641628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE transactions on pattern analysis and machine intelligence , volume=","venue":null,"work_id":"76471e93-a86b-4393-8173-73e5014962ff","year":2021},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:8801c06aa5574cf6a5c7fc722926050d6574e6d88d23bb0f9c0f9cee8c96cc6d","observation_id":"509e0e6e-f60c-4ce1-a9ba-11566927791b","resolution":{"observed_at":"2026-05-20T20:23:43.639796Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.02999","last_updated":"2018-10-22T16:11:14Z","snapshot_observed_at":"2026-08-05T12:53:31.632675Z","submitted_at":"2018-03-08T08:29:38Z","title":"On First-Order Meta-Learning Algorithms","version":3},"cited_work":{"arxiv_id":"1803.02999","doi":null,"metadata_source":"pith","pith_arxiv_id":"1803.02999","snapshot_observed_at":"2026-07-08T22:55:40.216179Z","title":"On First-Order Meta-Learning Algorithms","venue":"cs.LG","work_id":"cfd46fb8-7c98-41ac-8499-e11236b08f18","year":2018},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/1803.02999","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:a73b97d4fe3934c094c2fdfe6340fafbf684b1a93476a3d2dc3271a3f662997f","observation_id":"2df3e228-70cb-4a80-8f44-3187ac8101fd","resolution":{"observed_at":"2026-05-20T20:23:43.283425Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-08-02T01:40:54.187278Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:5aa386cb605b8936d8b2b79693097edb85cc9bb417af915876b2dd6a36550725","observation_id":"c108e4e6-efd1-4716-97d5-eaabab27e103","resolution":{"observed_at":"2026-05-20T20:23:43.235537Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":"2112.09332","doi":"10.48550/arxiv.2112.09332","metadata_source":"pith","pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebGPT: Browser-assisted question-answering with human feedback","venue":"cs.CL","work_id":"e25ef3e1-4848-4cb9-bf28-67a420591165","year":2021},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:a8c65aa30cf29d508c6fec083b04508f5ab3aec1faffe0716f89ce22a7ddad3a","observation_id":"1176fd6b-9a16-41f8-9334-7257b7525361","resolution":{"observed_at":"2026-05-20T20:23:43.310355Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T09:36:10.674129Z","title":"International conference on machine learning , pages=","venue":null,"work_id":"760b046d-85cf-4e84-908d-6bdc815674ad","year":2017},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:157b6f47c6bcd1ccfcb52ebf0a055103c6680f9633e4254144006df11a34c594","observation_id":"0395589b-2b48-4624-97e7-66e7201ca3f8","resolution":{"observed_at":"2026-05-20T20:23:43.637699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"11c71b9e-22a3-4589-ba6c-547dc0a42bab","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:63612db76c37ef8bdbd8cc0a325110e5ac81bcbbc8d91b6ef86aabd742db4181","observation_id":"76eed3a5-7e5a-4cb8-8725-a308cc4a8fff","resolution":{"observed_at":"2026-05-20T20:23:43.643491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.18071","last_updated":"2025-07-28T11:11:33Z","snapshot_observed_at":"2026-08-06T04:31:03.113409Z","submitted_at":"2025-07-24T03:50:32Z","title":"Group Sequence Policy Optimization","version":2},"cited_work":{"arxiv_id":"2507.18071","doi":"10.48550/arxiv.2507.18071","metadata_source":"pith","pith_arxiv_id":"2507.18071","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Group Sequence Policy Optimization","venue":"cs.LG","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2507.18071","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:a235f707e1e923ada7824d74c4ec3e11d5a47e0b47f3483fbbca3b08835dc7f3","observation_id":"656a3345-2b1a-44a7-97c9-33cd94bfb0d4","resolution":{"observed_at":"2026-05-20T20:23:43.182269Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02474","last_updated":"2026-05-24T19:01:09Z","snapshot_observed_at":"2026-08-03T05:25:38.558727Z","submitted_at":"2026-02-02T18:53:28Z","title":"MemSkill: Learning and Evolving Memory Skills for Self-Evolving Agents","version":2},"cited_work":{"arxiv_id":"2602.02474","doi":"10.48550/arxiv.2602.02474","metadata_source":"pith","pith_arxiv_id":"2602.02474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MemSkill: Learning and Evolving Memory Skills for Self-Evolving Agents","venue":"cs.CL","work_id":"64baa7f3-849b-485e-b081-2074d82f1364","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2602.02474","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:dea53ace382c4ac8b851170364501bfe0b97d1917a647956784b900e0743c944","observation_id":"5dd3e6b2-8875-4803-9d4f-1cb3b55e528f","resolution":{"observed_at":"2026-05-20T20:23:43.273474Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.02553","last_updated":"2026-01-29T15:45:24Z","snapshot_observed_at":"2026-07-06T22:40:46.465095Z","submitted_at":"2026-01-05T21:02:49Z","title":"SimpleMem: Efficient Lifelong Memory for LLM Agents","version":3},"cited_work":{"arxiv_id":"2601.02553","doi":"10.48550/arxiv.2601.02553","metadata_source":"pith","pith_arxiv_id":"2601.02553","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SimpleMem: Efficient Lifelong Memory for LLM Agents","venue":"cs.AI","work_id":"9b5b2649-8727-425e-a2c5-60f49518a5c4","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2601.02553","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:8d79019c00d9ca5b8c36133323c1c497de968874342a690286cf720fa730a8f7","observation_id":"f6ffcddd-0919-425d-a233-6a54e2861649","resolution":{"observed_at":"2026-05-22T08:10:45.412758Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.10538","last_updated":"2023-12-03T13:18:09Z","snapshot_observed_at":"2026-07-06T16:49:04.651019Z","submitted_at":"2023-11-17T14:06:05Z","title":"Testing Language Model Agents Safely in the Wild","version":3},"cited_work":{"arxiv_id":"2311.10538","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.10538","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2311.10538 , year=","venue":null,"work_id":"c9e321a2-db8b-455d-a971-def428e0e50d","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2311.10538","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:cf452ed8708eb4375dd299cb84b5e6b8aca89dd3b4429fb88b3dffa22489ff8d","observation_id":"c480a274-e797-43b3-a8ca-7a35bca8a40a","resolution":{"observed_at":"2026-05-20T20:23:43.230284Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"21d011bb-5c93-40a9-96b1-f208e66ba270","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:d3d7649350000507e72fe5bccc1bc7b9fefce1edc1c7efea969e60518d2803f9","observation_id":"f0aa1872-a5ff-4330-88c7-3eee64bbcc57","resolution":{"observed_at":"2026-05-20T20:23:43.635925Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.03312","last_updated":"2025-09-04T17:49:20Z","snapshot_observed_at":"2026-08-07T03:20:48.934367Z","submitted_at":"2025-09-03T13:42:14Z","title":"AgenTracer: Who Is Inducing Failure in the LLM Agentic Systems?","version":2},"cited_work":{"arxiv_id":"2509.03312","doi":"10.48550/arxiv.2509.03312","metadata_source":"pith","pith_arxiv_id":"2509.03312","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zhang, J","venue":"cs.CL","work_id":"bcc7b37c-db67-4d9b-8cc2-9458cc338abd","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2509.03312","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:64eb5d3438729c062a569bf3724ae95576dc485baf501fd1acbbd6788b711e6b","observation_id":"e7a1938f-aca4-4f4f-bf34-0c20b1df9780","resolution":{"observed_at":"2026-05-20T20:23:43.307883Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:55.524505+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:55.524505+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.08234","last_updated":"2026-02-09T03:17:17Z","snapshot_observed_at":"2026-07-06T22:45:05.823859Z","submitted_at":"2026-02-09T03:17:17Z","title":"SkillRL: Evolving Agents via Recursive Skill-Augmented Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2602.08234","doi":"10.48550/arxiv.2602.08234","metadata_source":"pith","pith_arxiv_id":"2602.08234","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SkillRL: Evolving Agents via Recursive Skill-Augmented Reinforcement Learning","venue":"cs.LG","work_id":"536a9d0a-baed-4eb9-9a51-f2b585c3388b","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2602.08234","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:a219ffebcf7b12df61ac801646542addc4a4ec046eb76140567d0dbade7a1757","observation_id":"80e583a1-106c-492b-b2ff-080ba6a5b621","resolution":{"observed_at":"2026-05-20T20:23:43.300937Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13564","last_updated":"2026-01-13T09:33:57Z","snapshot_observed_at":"2026-08-06T08:27:07.254588Z","submitted_at":"2025-12-15T17:22:34Z","title":"Memory in the Age of AI Agents","version":2},"cited_work":{"arxiv_id":"2512.13564","doi":"10.48550/arxiv.2512.13564","metadata_source":"pith","pith_arxiv_id":"2512.13564","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Memory in the Age of AI Agents","venue":"cs.CL","work_id":"1ff75a14-302e-4906-aa86-1f96fbcf12ff","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2512.13564","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:8e16d7f3e282441ab625fa4864b5a42faf33dcfbf972392668db603a7c162a47","observation_id":"2ea9022d-7cf4-4d9f-bce8-bcdd5975f5ad","resolution":{"observed_at":"2026-05-20T20:23:43.185292Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.25140","last_updated":"2026-03-16T20:49:28Z","snapshot_observed_at":"2026-08-02T12:08:17.149184Z","submitted_at":"2025-09-29T17:51:03Z","title":"ReasoningBank: Scaling Agent Self-Evolving with Reasoning Memory","version":2},"cited_work":{"arxiv_id":"2509.25140","doi":"10.48550/arxiv.2509.25140","metadata_source":"pith","pith_arxiv_id":"2509.25140","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ReasoningBank: Scaling Agent Self-Evolving with Reasoning Memory","venue":"cs.AI","work_id":"551218b8-a306-4c1e-8795-6232cc30192b","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2509.25140","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:b577f7d72d5ded0c639e962f4c6f99ef7f79f76ad812bf19e7fb0711a2284cc8","observation_id":"570e4bb4-7db3-4ad4-a78a-88c6062da656","resolution":{"observed_at":"2026-05-20T20:23:43.290691Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-22T15:52:34.995342+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-22T15:52:34.995342+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.06229","doi":"10.48550/arxiv.2507.06229","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agent kb: Leveraging cross-domain experience for agentic problem solving","venue":"ArXiv.org","work_id":"11232777-9790-41f1-ae97-5d0c1577d5a0","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:564a8f8b498463fec28558fc35a62e95e2d0feae2216a0101c78afc91f8a0444","observation_id":"ccb9b716-cdf5-4de6-ae34-619a021625fe","resolution":{"observed_at":"2026-05-20T20:23:43.242300Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:57:35.737441Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"44940506-60da-47fb-a734-bf5bba9cd5fd","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:d780b253f2b86128d4eabd08e7d57afdf166871e53a7b48809c026cf9476db34","observation_id":"8b8471d5-1755-4bce-992d-417782c54952","resolution":{"observed_at":"2026-05-20T20:23:43.630219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the 62nd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) , pages=","venue":null,"work_id":"510a9c09-bf64-4c56-ae84-7bf70f09357d","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:0cc66a85f04abe51539181335e2f080f1a88561eccb3b98b27e58a26283da734","observation_id":"589e422e-f02f-471d-9d21-4ddbe6f33688","resolution":{"observed_at":"2026-05-20T20:23:43.632124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:ff48e0f3f899bbe890f37a77a98b98f8d21b6bde80b1fd1cf9340031d27d9092","observation_id":"2989f9d4-7382-4301-891c-091ae6f1f227","resolution":{"observed_at":"2026-05-20T20:23:43.191304Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.16079","last_updated":"2026-05-16T02:57:47Z","snapshot_observed_at":"2026-08-05T00:54:36.690304Z","submitted_at":"2025-10-17T12:03:16Z","title":"EvolveR: Self-Evolving LLM Agents through an Experience-Driven Lifecycle","version":3},"cited_work":{"arxiv_id":"2510.16079","doi":"10.48550/arxiv.2510.16079","metadata_source":"pith","pith_arxiv_id":"2510.16079","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"EvolveR: Self-Evolving LLM Agents through an Experience-Driven Lifecycle","venue":"cs.CL","work_id":"350af93c-7749-4477-8163-e35d2cac8d96","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2510.16079","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:0d1b8611433b21bc403b3d4df41d675ec2556309d13e521ca1b56fef2914afd7","observation_id":"5d6359df-0977-4115-a374-4cb799265576","resolution":{"observed_at":"2026-05-20T20:23:43.312709Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.06433","last_updated":"2026-04-15T17:21:59Z","snapshot_observed_at":"2026-07-06T22:10:07.250122Z","submitted_at":"2025-08-08T16:20:56Z","title":"Memp: Exploring Agent Procedural Memory","version":4},"cited_work":{"arxiv_id":"2508.06433","doi":"10.48550/arxiv.2508.06433","metadata_source":"pith","pith_arxiv_id":"2508.06433","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Memp: Exploring Agent Procedural Memory","venue":"cs.CL","work_id":"7a9faca4-10b1-472d-8fc5-89b683d01851","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2508.06433","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:076626ad77fd66366a20e3127bbe4db883a3d9fb0874cbfd62408920c3607d9d","observation_id":"1c0bcbfc-01d5-4e1e-8ab1-378d8d9d23e7","resolution":{"observed_at":"2026-05-20T20:23:43.197029Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:68b68029af4a640427d3a36f0b026913911b24674c0a19f5a259aea7315d2279","observation_id":"d1141c40-cffb-498c-9546-d3f32f126f6d","resolution":{"observed_at":"2026-05-20T20:23:43.188388Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10978","last_updated":"2025-10-28T15:11:36Z","snapshot_observed_at":"2026-07-29T19:20:21.974239Z","submitted_at":"2025-05-16T08:26:59Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","version":3},"cited_work":{"arxiv_id":"2505.10978","doi":"10.48550/arxiv.2505.10978","metadata_source":"pith","pith_arxiv_id":"2505.10978","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","venue":"cs.LG","work_id":"bc65d492-e6ba-4522-874c-43d2f4fc5191","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2505.10978","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:ad268ab4e2eb21727c32335db479343db4ea255c5f9ce31c0feb34a211ea19ea","observation_id":"a1ccfae9-bab3-4898-93f1-3289fc5327f4","resolution":{"observed_at":"2026-05-20T20:23:43.211084Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2025 , author=","venue":null,"work_id":"159ed23a-305c-41ca-bc29-dc1d8c5cee0c","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:2abfdd8acba30460731b2223e73f9ab5fe31c9fce136a63787d4ada27bf0b240","observation_id":"0896cfae-2f57-47b8-97e2-77b05a8387e9","resolution":{"observed_at":"2026-05-20T20:23:43.624591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:3044dc1444f12610e65d6d1ed1c5b60af8e506312cbcfbb8c27f0c59516e7e04","observation_id":"66ec27b4-e54b-4a71-941d-e67d1dfd3472","resolution":{"observed_at":"2026-05-20T20:23:43.315717Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2024 , author=","venue":null,"work_id":"de4b3550-7389-49c8-8f68-dc521eb5ca2f","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:661214ec3d5f81a3e06971e505289e010a0a3bc463b91fde5714c534207adc17","observation_id":"0954337a-a67d-4eea-97f8-edae9dc98a5c","resolution":{"observed_at":"2026-05-20T20:23:43.620875Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2025 , author=","venue":null,"work_id":"6c89985e-cd3f-47ac-a24e-ebd7a3c83ecf","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:db2ab704e523f9c6c7c3e07df967a1c56e8c9bf2dc7220765b222bb99091bbba","observation_id":"193767dd-1662-4075-929c-f4675355fe6f","resolution":{"observed_at":"2026-05-20T20:23:43.563601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2025 , author=","venue":null,"work_id":"bd00b796-738f-4770-8d1d-10043ffa44a4","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:6e15676c28d6d89e25d1bd9cf6fdc7702ac33c59ea587809128505b34217c58a","observation_id":"5e10ad2d-a3ec-40a2-8227-d31963b42d11","resolution":{"observed_at":"2026-05-20T20:23:43.581238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2024 , author=","venue":null,"work_id":"b5f33689-c6c7-4a3b-8309-d633c4636857","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":81,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:f0dd23c7184fbb6a01279e76b2a238eec818adcbdedc48b9f27062c8a6ff8326","observation_id":"b503fd6e-69a1-43d2-b184-fa67ac06084a","resolution":{"observed_at":"2026-05-20T20:23:43.613147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.24701","last_updated":"2026-05-18T04:10:32Z","snapshot_observed_at":"2026-07-06T22:34:18.297603Z","submitted_at":"2025-10-28T17:53:02Z","title":"Tongyi DeepResearch Technical Report","version":3},"cited_work":{"arxiv_id":"2510.24701","doi":"10.48550/arxiv.2510.24701","metadata_source":"pith","pith_arxiv_id":"2510.24701","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Tongyi DeepResearch Technical Report","venue":"cs.CL","work_id":"1c8db01b-b50f-4711-b181-a04bcc3e9aa8","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2510.24701","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:255f83e6566676f7a43d7c3f96213ed1c83c279cabc5d91c1d50a0321b7ebdbd","observation_id":"078254ff-ecf8-4552-b090-1c1fc1e866db","resolution":{"observed_at":"2026-05-20T20:23:43.271148Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19413","last_updated":"2025-04-28T01:46:35Z","snapshot_observed_at":"2026-08-02T07:32:11.339534Z","submitted_at":"2025-04-28T01:46:35Z","title":"Mem0: Building Production-Ready AI Agents with Scalable Long-Term Memory","version":1},"cited_work":{"arxiv_id":"2504.19413","doi":"10.48550/arxiv.2504.19413","metadata_source":"pith","pith_arxiv_id":"2504.19413","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mem0: Building Production-Ready AI Agents with Scalable Long-Term Memory","venue":"cs.CL","work_id":"a5aed26c-a248-48b6-a59e-f7693fcb180a","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2504.19413","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:b06396fa2b739f3a7225b7dbc7b4b0967f679b654bcc37c7eafc74b5196dd81d","observation_id":"8e158ed5-693e-4b2c-a033-26d60cf63a92","resolution":{"observed_at":"2026-05-20T20:23:43.278367Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.01885","last_updated":"2026-07-23T06:23:36Z","snapshot_observed_at":"2026-08-03T14:57:06.604444Z","submitted_at":"2026-01-05T08:24:16Z","title":"Agentic Memory: Learning Unified Long-Term and Short-Term Memory Management for Large Language Model Agents","version":3},"cited_work":{"arxiv_id":"2601.01885","doi":"10.48550/arxiv.2601.01885","metadata_source":"pith","pith_arxiv_id":"2601.01885","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Agentic Memory: Learning Unified Long-Term and Short-Term Memory Management for Large Language Model Agents","venue":"cs.CL","work_id":"4519e90e-26c9-43ef-a89b-c8e8a8378280","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2601.01885","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:de3abdd0e6649d735c528d571ad55e08e3e8798c205d821f96edd481866bd21c","observation_id":"093e8c82-8f7e-418b-9b27-d9c567c06f94","resolution":{"observed_at":"2026-05-20T20:23:43.238254Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T22:35:40.604854Z","title":"Findings of the Association for Computational Linguistics: ACL 2024 , pages=","venue":null,"work_id":"47a0bced-d587-4ffc-8e94-107a6c09d198","year":2024},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:4bab944a13ba72cccf3fc09385cb8e81668b9a7951720109b3c2163da1ad22b9","observation_id":"0c003b22-810e-46b5-85f5-02c42d3b06d0","resolution":{"observed_at":"2026-05-20T20:23:43.626388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:a293f8a1074cc8092e429a06033f8088fa3766f414e62a0fbb1d9b047a326e3d","observation_id":"eb5338d2-f0a9-41a7-bb42-4b6fad8d1578","resolution":{"observed_at":"2026-05-20T20:23:43.208503Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05915","last_updated":"2023-10-09T17:58:38Z","snapshot_observed_at":"2026-08-05T18:32:49.850038Z","submitted_at":"2023-10-09T17:58:38Z","title":"FireAct: Toward Language Agent Fine-tuning","version":1},"cited_work":{"arxiv_id":"2310.05915","doi":"10.48550/arxiv.2310.05915","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.05915","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Fireact: Toward language agent fine-tuning","venue":"arXiv (Cornell University)","work_id":"382904a8-a33f-42d0-ae77-b61c9f0cb7cb","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2310.05915","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:3767ed97904f6243a8477cab710c731326d520b46941abcea4c0586a96202d3a","observation_id":"03da1012-3696-4a93-90a2-e6fa14d1f4d9","resolution":{"observed_at":"2026-05-20T20:23:43.246128Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:31bfd396ebf2f5e48d75496591005727a748c29ce9b2d0251b4f8a28b298d440","observation_id":"b6e3a6a4-92ad-4296-bd33-8f6fb192b888","resolution":{"observed_at":"2026-05-20T20:23:43.227288Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:42:56.204813Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"8e05574b-4459-477c-8478-5f60a3bb36a2","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:2fa9c0a6ceeb2950704af21f28b6cfd7f6e5d0c3197487aa7c652f8c81772429","observation_id":"ef44d215-85a0-4fbc-95f9-19c312d1161a","resolution":{"observed_at":"2026-05-20T20:23:43.605880Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"9a6b3329-d81b-4b25-a8d2-d4d18dd8cc4c","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:c15147d8b3774ae1160fb28e46407738e7d151cad1a8b591d178a4172990a184","observation_id":"bd352baf-f58a-421a-a73e-8fe55beb5764","resolution":{"observed_at":"2026-05-20T20:23:43.561919Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2023 IEEE International Conference on Robotics and Automation (ICRA) , pages=","venue":null,"work_id":"b4c38574-e8dd-4910-bfbf-b8a1877b6b93","year":2023},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:26d2b76e115a337db599ea91ce63831be097127c4e765f12e1306cb6dc707641","observation_id":"3f53375a-8356-48f2-b947-a85ca6ede89e","resolution":{"observed_at":"2026-05-20T20:23:43.602470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"9642fcf4-8a2d-44c8-b043-64c4468a1b59","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:c74863e5acfd075e746f32011e9d96238418a295e7a16febbf5a2d906c5b8e45","observation_id":"7b6f92bc-4103-4c0e-99b1-1c25a22b2b70","resolution":{"observed_at":"2026-05-20T20:23:43.604189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T06:43:30.619307Z","title":"Artificial intelligence , volume=","venue":null,"work_id":"fa8ffd0d-696d-4d4d-86f1-83c82c56f23e","year":1999},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:937597e9ed468cf06a404b21c0e6d910cbc34c6c8bc7a2fe0d67cccfd3bd8b78","observation_id":"201985b5-69db-472d-8c9b-d4c3abd33e7a","resolution":{"observed_at":"2026-05-20T20:23:43.597041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T21:52:55.769480Z","title":"Proceedings of the AAAI Conference on Artificial Intelligence , volume=","venue":null,"work_id":"8d804125-92d2-42ea-908c-dc36c3d8e41f","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:3579089db639433d4b35ed9be88fab0e2a959fa4890960a700fb2ee6ff0e3801","observation_id":"2b626a41-123d-40dc-be3b-156607899dfb","resolution":{"observed_at":"2026-05-20T20:23:43.560119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"10addd71-9ca0-45d0-acbb-de0eb3918238","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:37bc16e0b9c319c3d305d1e0eb0463d4c42b96d222aa20686efb06dea4730cd4","observation_id":"7268c921-ab63-407b-86fe-f30b413d5e4d","resolution":{"observed_at":"2026-05-20T20:23:43.599011Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T20:57:35.726030Z","title":"First Conference on Language Modeling , year=","venue":null,"work_id":"9f93e065-7c5f-4d0b-9f39-cbbfbcc40246","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:fd26009885676a68881a2c376f30798be80965aaf92d24ecfd8234057501eff2","observation_id":"ec6519f3-ab46-4151-aca2-fd4c2f6d94e7","resolution":{"observed_at":"2026-05-20T20:23:43.558320Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T07:03:26.855360Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"5dda28f0-c02b-4e75-afa2-b5c5b3aa994e","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:80f3535bc26b512c6a5ed4e4114e8c1bc55d20c364a5640903387fe41502bab0","observation_id":"d80640d4-a9b6-482a-83d1-d9abae162931","resolution":{"observed_at":"2026-05-20T20:23:43.633974Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T15:05:03.646026Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":"a2bdafeb-48d9-47c6-8520-fad9a5a3c738","year":null},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:aa49337be41cf034268ec46e6ebfe3f2d5cf4b92ab85b6b425cdc95a36de08c1","observation_id":"512679d3-2900-45f1-9bb9-b348ed991bbf","resolution":{"observed_at":"2026-05-20T20:23:43.594966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.07957","last_updated":"2025-07-10T17:40:11Z","snapshot_observed_at":"2026-08-06T11:34:56.054509Z","submitted_at":"2025-07-10T17:40:11Z","title":"MIRIX: Multi-Agent Memory System for LLM-Based Agents","version":1},"cited_work":{"arxiv_id":"2507.07957","doi":"10.48550/arxiv.2507.07957","metadata_source":"pith","pith_arxiv_id":"2507.07957","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MIRIX: Multi-Agent Memory System for LLM-Based Agents","venue":"cs.CL","work_id":"bcb99885-6f76-4bd6-b442-e5f455bb6447","year":2025},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2507.07957","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:b2d26133f2c2517ea73bf728c153bf0b374ffd581a99bb152f683cba8918a200","observation_id":"dead71bf-aece-4abb-8e00-e3923bc08082","resolution":{"observed_at":"2026-05-20T20:23:43.222323Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.12538","last_updated":"2026-01-18T18:58:23Z","snapshot_observed_at":"2026-08-04T22:42:23.171653Z","submitted_at":"2026-01-18T18:58:23Z","title":"Agentic Reasoning for Large Language Models","version":1},"cited_work":{"arxiv_id":"2601.12538","doi":"10.48550/arxiv.2601.12538","metadata_source":"pith","pith_arxiv_id":"2601.12538","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agentic Reasoning for Large Language Models","venue":"cs.AI","work_id":"062546cd-e1a7-46e4-b617-c6b8e19b6fa3","year":2026},"citing_paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","version":2},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-05-20T20:19:21.824216Z"},"links":{"cited_paper":"/paper/2601.12538","citing_paper":"/paper/2605.14133"},"observation_digest":"sha256:05e39cdfb867d1ad741753f374d2552f2b2ac9fd495563b465de31fed2bed362","observation_id":"0959e00a-2a79-4d39-bbed-43218083d110","resolution":{"observed_at":"2026-05-20T20:23:43.249036Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.217033+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.217033+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.14133","last_updated":"2026-05-18T05:36:27Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-07-06T23:25:34.835136Z","submitted_at":"2026-05-13T21:34:08Z","title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":48,"parse_uncertain":1,"unresolved":4,"verified_exact":3,"verified_fuzzy":44},"total_outbound_references":101},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 101 outbound references and 1 inbound Pith citation observation for arXiv:2605.14133."}