{"as_of":"2026-08-05T21:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fd6aa5bd350de303ae4b80fd4e83f94ffa90727f43c08afacfffde6bb55a1574","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T09:56:36.860369Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.03889/citation-record","integrity":"/paper/2606.03889/integrity","json":"/paper/2606.03889/citation-record.json","paper":"/paper/2606.03889"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:e12ff335bc9c74e5d49fc0ce8039b84f2ef7f773c16544dc19c4f8d803104943","observation_id":"47f26d2e-c721-483f-98e1-a46d562044e4","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.27922","last_updated":"2026-05-27T03:47:35Z","snapshot_observed_at":"2026-08-05T01:33:47.362496Z","submitted_at":"2026-05-27T03:47:35Z","title":"Harness-Bench: Measuring Harness Effects across Models in Realistic Agent Workflows","version":1},"cited_work":{"arxiv_id":"2605.27922","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.27922","snapshot_observed_at":"2026-07-09T23:26:36.859251Z","title":"Harness-Bench: Measuring Harness Effects across Models in Realistic Agent Workflows","venue":"cs.AI","work_id":"20fe2270-cc99-4019-8456-7de7d0a81c8a","year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2605.27922","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:83ff33e3de4285c973ff5f629489f72a07bc1128f8bf58736bd7df6211d168a3","observation_id":"a0b0d40a-e0c0-4065-9be1-386ae646d9a6","resolution":{"observed_at":"2026-07-02T03:36:29.209900Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.12730","doi":"10.48550/arxiv.2512.12730","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nl2repo-bench: Towards long-horizon repository generation evaluation of coding agents.CoRR, abs/2512.12730","venue":"ArXiv.org","work_id":"0ea67a78-d9a4-4898-ab97-00d13ebca55f","year":2025},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:5afed38d4c9d0284834161f7606f4284224d815614ed78c539a73baa1ec2e552","observation_id":"4c98d1ac-88b0-427d-9de9-117041e52c48","resolution":{"observed_at":"2026-07-02T03:36:29.227775Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":"2112.09332","doi":"10.48550/arxiv.2112.09332","metadata_source":"pith","pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebGPT: Browser-assisted question-answering with human feedback","venue":"cs.CL","work_id":"e25ef3e1-4848-4cb9-bf28-67a420591165","year":2021},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:f3852b7868e132fe76e747ca58eaf960abf8f81b1af225f61e27fdd37f603620","observation_id":"e978255a-0166-4534-bc3a-ca14b500f343","resolution":{"observed_at":"2026-07-02T03:36:29.198722Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.00445","last_updated":"2022-05-01T11:01:28Z","snapshot_observed_at":"2026-07-06T13:05:30.057038Z","submitted_at":"2022-05-01T11:01:28Z","title":"MRKL Systems: A modular, neuro-symbolic architecture that combines large language models, external knowledge sources and discrete reasoning","version":1},"cited_work":{"arxiv_id":"2205.00445","doi":"10.48550/arxiv.2205.00445","metadata_source":"pith","pith_arxiv_id":"2205.00445","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MRKL Systems: A modular, neuro-symbolic architecture that combines large language models, external knowledge sources and discrete reasoning","venue":"cs.CL","work_id":"e393ecfc-aa97-4014-bb7a-90ebd9f00535","year":2022},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2205.00445","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:0926bfa2920e4d91737f6201e2a08c45cab40223667777dc6157fff5d771cad3","observation_id":"9bed8443-d1b7-470b-be08-bdae4b54e9ff","resolution":{"observed_at":"2026-07-02T03:36:29.230355Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:88f8307d2a1aca7945c649056ade1a5a2e58a76faaa4e496f3550497da50cbeb","observation_id":"14da92cb-f061-485a-92c8-a14371b7b22e","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":"2210.03629","doi":"10.48550/arxiv.2210.03629","metadata_source":"pith","pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","venue":"cs.CL","work_id":"407a2351-25f1-497d-b611-f77d0292a8e6","year":2022},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:1cc8ec2dee29bae1cf49b521048d9eff7ee2cf2deccc7314e34244459f53a03c","observation_id":"ec76bab7-02c4-474c-8788-f82b8713a307","resolution":{"observed_at":"2026-07-02T03:36:29.221727Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:36.897515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:36.897515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:a00bdae8877720e26956f5273c014f8a59528f27f5467f1411d3c00a03a05241","observation_id":"62a60cf3-dbdf-480e-8c54-8f1e9b16f5f0","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:5905ba383ecf00bb576060014c2773ae3050e221028451680eaa22314103541e","observation_id":"749f623c-be79-4409-9b38-c5c89f3f16db","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:d317a6e823787b4a2a95a3f7f9236c227bc6c8a76ddc1c8690bf7018f3eb0d4f","observation_id":"2a2fe6ba-958b-4b6b-84e4-46de76490b88","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:0fd0dddf81c2b3bcf700af80976ca376556e4ff789fbd7e5ad2480c5cb731dce","observation_id":"61f39f04-a20e-4f04-89ee-b9bf037aea22","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:c70e7ae584653a736ca0ed2cf60a55e522d7a2b27d7539d9ddc45aaaedc029bd","observation_id":"154624c0-998a-4213-b837-63a1e26229ca","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:1178500cc04f5c585fafbe17702b422089be12d506a65521af1d68bbc5cb17ed","observation_id":"caa83049-8b36-4d3f-ac93-184492759558","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"Advances in neural information processing systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:ad52a79628cbdca7ec0bc3350eab09260a013b2f27453767a9039ae931525377","observation_id":"250c8be5-2c09-4b44-afcf-9a1d26e6a06c","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":"2009.03300","doi":"10.48550/arxiv.2009.03300","metadata_source":"pith","pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Massive Multitask Language Understanding","venue":"cs.CY","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","year":2020},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:336e9259c4adc2c00296369ba84792e5433d5adf7116167177a5cd153f903b21","observation_id":"be860491-95c7-4319-b795-00da4020a766","resolution":{"observed_at":"2026-07-02T03:36:29.219031Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:06.256034+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"Transactions on machine learning research , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:4d44a713bb8f849e4c13a47a77a0d7f360da9a7eb724465b8d11ebacbb81ff11","observation_id":"54c760d9-80c6-4106-9882-efd73cd98791","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-01T19:14:56.803459Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":"2211.09110","doi":"10.1007/bf01194075","metadata_source":"pith","pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Holistic Evaluation of Language Models","venue":"cs.CL","work_id":"cc02a01e-7218-47dc-8e66-3333e7e4adec","year":2022},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:4f107939a23888f1f29610bd16a45e4a9f5bd7dd5d2654c74467d4f55a954c9e","observation_id":"56d720e4-c726-4553-9634-abc3979cb863","resolution":{"observed_at":"2026-07-02T03:36:29.212752Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:e808a89dd1558da668d19c1150c371cfc6d6a91a9872c9f4b5fc90705b1fb14a","observation_id":"527a12fe-6549-4a1c-a29f-b7743d5d2781","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"Proceedings of the 2023 conference on empirical methods in natural language processing , pages=","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:9435251b089a5b5c2c3aad18c77125d2d423164ac518c36e9835529897af2a67","observation_id":"0a1c4164-8d92-40a0-b82e-92a201887d43","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"Findings of the Association for Computational Linguistics: ACL 2024 , pages=","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:39df5daa19fab363ebee97b252a86db32f6493bb096997d67178bfdc7a6b779b","observation_id":"aad151d5-8481-4196-a147-d3d29a8c33a4","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-02T22:19:29.043854Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:e6c06b858974a9c7d9b50cf528b989a48289b1f6dc3cab6e4dc7c6cfc59cd697","observation_id":"69f44750-5152-445e-8e10-3247cc16d2d2","resolution":{"observed_at":"2026-07-02T03:36:29.224672Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"Advances in Neural Information Processing Systems , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:c2655ad9d47d1a96442077db6ade04de77af686f2495cb39a39e4bd51d4ea0ed","observation_id":"a270ea03-842d-4e9c-8cfd-65c712c02b10","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07718","last_updated":"2024-07-23T06:19:28Z","snapshot_observed_at":"2026-08-02T12:09:24.340284Z","submitted_at":"2024-03-12T14:58:45Z","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","version":5},"cited_work":{"arxiv_id":"2403.07718","doi":"10.48550/arxiv.2403.07718","metadata_source":"pith","pith_arxiv_id":"2403.07718","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","venue":"cs.LG","work_id":"5ac27d9e-4522-46f8-985e-0e4f73130803","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2403.07718","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:57a9b7d42bf1479dbd7f22e93b400b4486d6c9dc56d4a7833b910aa3cb66af1d","observation_id":"5529de73-394b-437c-9741-b663681d62fd","resolution":{"observed_at":"2026-07-02T03:36:29.201609Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:646760fc62fdb505724fc4b2c8a2166a1682a0f5f989371baf987a9c73187f0b","observation_id":"bb2696f8-df19-4e7c-90af-a0223aacb4d4","resolution":{"observed_at":"2026-07-02T03:36:29.215914Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:cd8ad3eee72072c8cbf09e8c381fcb133d6d306be2dfc144345ac7176fb416a6","observation_id":"a571dc55-908d-4f6d-97d5-6610a07f3507","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04132","last_updated":"2024-03-07T01:22:38Z","snapshot_observed_at":"2026-08-02T17:55:33.750637Z","submitted_at":"2024-03-07T01:22:38Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","version":1},"cited_work":{"arxiv_id":"2403.04132","doi":"10.1007/s11336-009-9136-x","metadata_source":"pith","pith_arxiv_id":"2403.04132","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Chatbot Arena: An Open Platform for Evaluating LLMs by Human Preference","venue":"cs.AI","work_id":"a1eb83da-5727-4b63-977f-81c6fdd33936","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2403.04132","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:89bd388a5b26c2d0486d64535d83cd8b9a4e64532f3e2f1d5955af64de938c7f","observation_id":"915bf9c3-9d18-4d79-9a82-343735faa7cf","resolution":{"observed_at":"2026-07-02T03:36:29.207143Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01470","last_updated":"2024-05-02T17:00:02Z","snapshot_observed_at":"2026-07-06T18:08:56.480769Z","submitted_at":"2024-05-02T17:00:02Z","title":"WildChat: 1M ChatGPT Interaction Logs in the Wild","version":1},"cited_work":{"arxiv_id":"2405.01470","doi":"10.48550/arxiv.2405.01470","metadata_source":"pith","pith_arxiv_id":"2405.01470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WildChat: 1M ChatGPT Interaction Logs in the Wild","venue":"cs.CL","work_id":"799d0d7b-7d66-40bb-b17f-205e5d2f3e13","year":2024},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2405.01470","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:991cb3349d6a709f24e3c6a94a2b184013133d06cedf57578a471cc71759e539","observation_id":"a13cb182-420a-4e03-bb20-5bbe3be840e6","resolution":{"observed_at":"2026-07-02T03:36:29.192106Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:547015d6b6bcfba51f3ce16fd41bda4ea27d273f4fc63bd599bdb7a691412ad9","observation_id":"83b2488c-4abf-4395-abd6-d3a2df855655","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.10912","last_updated":"2026-05-11T17:49:43Z","snapshot_observed_at":"2026-08-03T00:15:01.823825Z","submitted_at":"2026-05-11T17:49:43Z","title":"WildClawBench: A Benchmark for Real-World, Long-Horizon Agent Evaluation","version":1},"cited_work":{"arxiv_id":"2605.10912","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.10912","snapshot_observed_at":"2026-07-11T01:37:42.499659Z","title":"WildClawBench: A Benchmark for Real-World, Long-Horizon Agent Evaluation","venue":"cs.CL","work_id":"070744ec-8dfb-4b87-96fe-d67e9e2114e8","year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2605.10912","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:7aeae6fa81fa41ecc198174e7f122e369b7bc882e1acbc389073a16804007044","observation_id":"2c37ca01-bdce-428f-8852-a5a0f4fd61df","resolution":{"observed_at":"2026-07-02T03:36:29.195665Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"2026 , url =","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:a9d8440cf6e3187eaf429fd60e6fd94277bcbcef794ee30239b9ccefd4ad75c5","observation_id":"c2afdec3-6c23-4bdf-80b7-927a2321f939","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.04759","last_updated":"2026-04-06T15:27:05Z","snapshot_observed_at":"2026-07-06T22:53:37.357996Z","submitted_at":"2026-04-06T15:27:05Z","title":"Your Agent, Their Asset: A Real-World Safety Analysis of OpenClaw","version":1},"cited_work":{"arxiv_id":"2604.04759","doi":"10.48550/arxiv.2604.04759","metadata_source":"pith","pith_arxiv_id":"2604.04759","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Your Agent, Their Asset: A Real-World Safety Analysis of OpenClaw","venue":"cs.CR","work_id":"210c09c9-0953-4b01-aed7-c4f462b807be","year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2604.04759","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:c4ffad642ef41d34749b4bd83098ce5471df1fbc6eae0d8487ceb0136c5b5a07","observation_id":"97b307b2-1370-423b-8aba-d86c2f2bcc1c","resolution":{"observed_at":"2026-07-02T03:36:29.204275Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.14858","last_updated":"2026-04-29T07:22:20Z","snapshot_observed_at":"2026-08-02T06:04:59.545465Z","submitted_at":"2026-04-16T10:45:13Z","title":"Benchmarks for Trajectory Safety Evaluation and Diagnosis in OpenClaw and Codex: ATBench-Claw and ATBench-Codex","version":2},"cited_work":{"arxiv_id":"2604.14858","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.14858","snapshot_observed_at":"2026-07-02T03:36:29.187827Z","title":"Benchmarks for Trajectory Safety Evaluation and Diagnosis in OpenClaw and Codex: ATBench-Claw and ATBench-Codex","venue":"cs.AI","work_id":"08d81a20-c784-4e07-b5c9-b38726eb6de1","year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2604.14858","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:28ff25834f1119280436795599d8c52f76104377b9afdd74294136ca2506e914","observation_id":"affb1a11-98fa-4243-8ff7-676e5c90f23e","resolution":{"observed_at":"2026-07-02T03:36:29.189095Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"2026 , howpublished=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:d4372851bc2937f816358bebaf5d4de2e9bae3f0ea15d1850a33273cc12d477b","observation_id":"b219f3ca-5021-49bf-8ba5-551e174b884a","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:5bb770422e8c4d154d733184f1c1f3167af7b8376195113a0c58d4a3d82bd894","observation_id":"eb957fc8-5941-4b76-86eb-9e083f5a9e4e","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"parse_uncertain"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.02113","last_updated":"2026-06-01T11:45:50Z","snapshot_observed_at":"2026-07-06T23:42:35.063977Z","submitted_at":"2026-06-01T11:45:50Z","title":"A Primer in Post-Training Reasoning Data: What We Know About How It Works","version":1},"cited_work":{"arxiv_id":"2606.02113","doi":"10.48550/arxiv.2606.02113","metadata_source":"pith","pith_arxiv_id":"2606.02113","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"A Primer in Post-Training Reasoning Data: What We Know About How It Works","venue":"cs.CL","work_id":"5bfd3c0c-f238-46ea-a18f-3c1bd02e8077","year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"cited_paper":"/paper/2606.02113","citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:1975d5e8b1962808c96435366cb5f9207877a4238dec70b328ecc384550d2459","observation_id":"7c4c5b44-b8a2-4d3f-9901-babc99db77f9","resolution":{"observed_at":"2026-06-28T10:01:52.590703Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:62bec2260b39113b378136604c7408c16e8d2af5f8caa294df892d3538728699","observation_id":"ac24918a-1eb6-4d4b-b631-399fb2313c4d","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T09:56:36.860369Z","title":"2026 , eprint=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-06-28T09:56:36.860369Z"},"links":{"citing_paper":"/paper/2606.03889"},"observation_digest":"sha256:453277a61654d8bd7a76fbca3f9fe4c2a43488b8d48417cf856a7e362419feba","observation_id":"e323426e-ba07-4d50-bb84-cc3e95966a04","resolution":{"observed_at":"2026-06-28T09:56:36.860369Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.03889","last_updated":"2026-06-05T09:38:21Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-02T13:33:31.496138Z","submitted_at":"2026-06-02T16:51:24Z","title":"RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":0,"metadata_mismatch":16,"parse_uncertain":2,"unresolved":19,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 0 inbound Pith citation observations for arXiv:2606.03889."}