{"as_of":"2026-08-07T09:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c862647184d6227e036409d0ddfc7f66d2848d64a88903be4108896a1800071d","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:47:23.367773Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.07441/citation-record","integrity":"/paper/2507.07441/integrity","json":"/paper/2507.07441/citation-record.json","paper":"/paper/2507.07441"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:19.026216Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.026216Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:26d2088290026abf9e60e90443424b3b6f8986fa929864bcc6e73f9300a698e7","observation_id":"d1c58432-c1bf-4e9d-9d6d-d27380de17db","resolution":{"observed_at":"2026-08-06T18:47:19.026216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:19.120230Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.120230Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:2293979bdbe7964499ecfdd860b008eef76a3e71574581281aa5aa03d96ed820","observation_id":"b1345c0c-97ec-4bdb-ba76-79784be37139","resolution":{"observed_at":"2026-08-06T18:47:19.120230Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T18:47:19.247638Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.247638Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:ee13ddb671ba1dfb248f71ac65307e8c7dc0c28a0b9b9abb4a275a55f790c428","observation_id":"1f951891-0a63-4b3b-98d3-0325132dc1e0","resolution":{"observed_at":"2026-08-06T18:47:19.247638Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05915","last_updated":"2023-10-09T17:58:38Z","snapshot_observed_at":"2026-08-05T18:32:49.850038Z","submitted_at":"2023-10-09T17:58:38Z","title":"FireAct: Toward Language Agent Fine-tuning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05915","snapshot_observed_at":"2026-08-06T18:47:19.417248Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.417248Z"},"links":{"cited_paper":"/paper/2310.05915","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:4a4165b41e658906e2f7e4c55411a04952705b293605f2406dc8670bd332dc30","observation_id":"4ece5b4b-0d92-46bc-a299-ae61de169c11","resolution":{"observed_at":"2026-08-06T18:47:19.417248Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.142316Z","title":null,"venue":null,"work_id":"db18f7e9-b782-41d0-83e6-0b731263929c","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.555232Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:d7680249fc6b537d59375b3153cad3cd221ce1c8ea508b5c07273bda9bd8a7e2","observation_id":"f1b399fd-71c6-495a-a48b-3b945872a3c1","resolution":{"observed_at":"2026-08-06T18:47:24.145308Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02197","last_updated":"2025-06-05T01:42:08Z","snapshot_observed_at":"2026-07-06T20:46:09.097525Z","submitted_at":"2025-03-04T02:14:55Z","title":"ATLaS: Agent Tuning via Learning Critical Steps","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02197","snapshot_observed_at":"2026-08-06T18:47:19.688593Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.688593Z"},"links":{"cited_paper":"/paper/2503.02197","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:a4b7bd72bbfe3f8ea280de065c4dd5670622ec09afb13571a1be56670835e7fc","observation_id":"febc73ab-b7d9-42e1-9fb6-6c1cc19d2ae6","resolution":{"observed_at":"2026-08-06T18:47:19.688593Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T18:47:19.800141Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.800141Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:5e07073db7925ae5920a38b612db332f952b4e1b9966ccadea80de40b23b28cc","observation_id":"2c1495a5-144f-40fd-bcbb-48e8e818efe4","resolution":{"observed_at":"2026-08-06T18:47:19.800141Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16339","last_updated":"2025-01-08T20:11:59Z","snapshot_observed_at":"2026-07-06T20:11:15.503499Z","submitted_at":"2024-12-20T21:00:11Z","title":"Deliberative Alignment: Reasoning Enables Safer Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16339","snapshot_observed_at":"2026-08-06T18:47:19.985383Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:19.985383Z"},"links":{"cited_paper":"/paper/2412.16339","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:f7c0a664ea097b166ed49f48ed36e2fb2e9c77a8f6b5cd02f62c5ced5c5c7d1a","observation_id":"e922bb64-c738-4675-aed0-39776862d27a","resolution":{"observed_at":"2026-08-06T18:47:19.985383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.11143","last_updated":"2025-10-09T12:22:46Z","snapshot_observed_at":"2026-07-31T12:28:37.704994Z","submitted_at":"2024-05-20T01:04:40Z","title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.11143","snapshot_observed_at":"2026-08-06T18:47:20.145227Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.145227Z"},"links":{"cited_paper":"/paper/2405.11143","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:59d2499e05481195d7b0f469623c9ab64de5d67c0111d2a67cd519fdc487febc","observation_id":"d479af49-068f-4c8a-b185-056e521f23de","resolution":{"observed_at":"2026-08-06T18:47:20.145227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:20.251707Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.251707Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:ee867afe4c6da4dd261a0957f60ee48b76a39235d788dd0616b4bfb291582ff3","observation_id":"83ca68c7-130b-41d1-95ca-d456ecddac34","resolution":{"observed_at":"2026-08-06T18:47:20.251707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.14507","last_updated":"2024-09-18T09:25:20Z","snapshot_observed_at":"2026-07-06T18:49:10.964379Z","submitted_at":"2024-07-19T17:59:03Z","title":"Internal Consistency and Self-Feedback in Large Language Models: A Survey","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.14507","snapshot_observed_at":"2026-08-06T18:47:20.378177Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.378177Z"},"links":{"cited_paper":"/paper/2407.14507","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:c061ce13b6f5ee2918979f123a666663d7d187e520001454793057d69f5a78cc","observation_id":"9c83b9cf-8803-4826-b029-1e45f8a4ea5d","resolution":{"observed_at":"2026-08-06T18:47:20.378177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.02584","last_updated":"2025-02-04T18:58:31Z","snapshot_observed_at":"2026-08-05T10:59:28.485092Z","submitted_at":"2025-02-04T18:58:31Z","title":"QLASS: Boosting Language Agent Inference via Q-Guided Stepwise Search","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.02584","snapshot_observed_at":"2026-08-06T18:47:20.506045Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.506045Z"},"links":{"cited_paper":"/paper/2502.02584","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:825d3638d331838a87cd5a5f495246d5fb4ebd3f2d9af0d19562fe0a025f961e","observation_id":"bcac6f08-a539-4ce4-a1a4-233b120afa61","resolution":{"observed_at":"2026-08-06T18:47:20.506045Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:20.615254Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.615254Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:f36933dae607fce7cf0a7775aed835fb441b4a10280eac66d4f35bba4287c29d","observation_id":"dfeec77f-50a9-43e2-98f4-73c923cc7f89","resolution":{"observed_at":"2026-08-06T18:47:20.615254Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-06T18:47:20.674156Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.674156Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:ee44edba7f09177d21923858ff0bcef4e9b3ca848a29da31b840a4ee94a1e3ca","observation_id":"2a1dce29-d0dc-4b0b-ad32-2a4042ff241d","resolution":{"observed_at":"2026-08-06T18:47:20.674156Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.120202Z","title":null,"venue":null,"work_id":"6f691064-736c-4a62-a423-653b98d5a9d9","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.801364Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:e426a047b505e21dd1243fa4e4d36a99a00df3b71fc325e9ce1cd868790d50bf","observation_id":"0e540441-45df-4b22-9e46-e1c6f479d971","resolution":{"observed_at":"2026-08-06T18:47:24.123063Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.emnlp-main.138","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:23.496439Z","title":null,"venue":null,"work_id":"aade76b0-2578-4e3f-867f-b1b181145384","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.899057Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:17185cac22f59d930b8bd5d5719358629c093f79871ce59e652ce6816306e6d2","observation_id":"b9d1d53f-8349-4011-adc4-2a88be6ebd52","resolution":{"observed_at":"2026-08-06T18:47:23.561522Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:20.991436Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:20.991436Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:ff27317262bba34a7ba493396e7e5df28c2168bae314a76e28ebaec0fd35b4c7","observation_id":"e1f0d83a-7e41-47e9-bcfb-b47baaca3f27","resolution":{"observed_at":"2026-08-06T18:47:20.991436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.03768","last_updated":"2021-03-14T22:44:38Z","snapshot_observed_at":"2026-07-06T10:02:33.297722Z","submitted_at":"2020-10-08T05:13:36Z","title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.03768","snapshot_observed_at":"2026-08-06T18:47:21.086928Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.086928Z"},"links":{"cited_paper":"/paper/2010.03768","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:9913f461661bad3321b2bcc06654fb5528c433e775d9e90efab7e9d60d84abe9","observation_id":"e8dd96be-fad7-46d0-b042-04279887dcd1","resolution":{"observed_at":"2026-08-06T18:47:21.086928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.103707Z","title":null,"venue":null,"work_id":"942c58b7-b6bb-42c6-9276-17178bb80712","year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.253533Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:f294e240aa6fb562a3cff477afe631b8ef5ff89bafab87ca3892246119c3bc61","observation_id":"ad2e3530-7d07-41c3-a562-a53f5142c921","resolution":{"observed_at":"2026-08-06T18:47:24.106815Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:21.432756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.432756Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:d09debc38537d9b2eb5b751a1effca58f0758ddf927d429ec4fa4a74cba59508","observation_id":"7d8b0230-658b-4639-bf03-1e40356300d7","resolution":{"observed_at":"2026-08-06T18:47:21.432756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.094537Z","title":null,"venue":null,"work_id":"a7dccfa6-ad60-4fc5-a8ab-4ce6abde3f31","year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.561733Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:8293567f78e00c96b8ddcfe5e99e26c92b7a50b4b80cc3fb02b96b01f660d7d8","observation_id":"165adf27-7a16-4ede-b22a-c2a605c72513","resolution":{"observed_at":"2026-08-06T18:47:24.097308Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.07540","last_updated":"2022-11-14T17:52:27Z","snapshot_observed_at":"2026-08-06T07:25:42.562786Z","submitted_at":"2022-03-14T22:52:34Z","title":"ScienceWorld: Is your Agent Smarter than a 5th Grader?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.07540","snapshot_observed_at":"2026-08-06T18:47:21.659567Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.659567Z"},"links":{"cited_paper":"/paper/2203.07540","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:ad148d74049248c7419602115158797aff51e68a09d59670505b9c3dbe11e3d3","observation_id":"1cad929e-0d5f-4796-9417-4ee8127b40f1","resolution":{"observed_at":"2026-08-06T18:47:21.659567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:21.809976Z","title":"Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.809976Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:fdad732d4686d4b65c9bcfe51af8d3c58d7af1d4242d768cdb81cfb8ebcd3695","observation_id":"a6362ff3-168c-41a8-a62c-e50eb85720ec","resolution":{"observed_at":"2026-08-06T18:47:21.809976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:21.904965Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:21.904965Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:e508cf313145b3a47c367fdcc303abb3a70c653739ad181149c751547219d9dc","observation_id":"f015df02-002d-4ea6-832d-a0b9e3e87cd4","resolution":{"observed_at":"2026-08-06T18:47:21.904965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18407","last_updated":"2025-02-25T17:58:02Z","snapshot_observed_at":"2026-07-06T20:42:33.518153Z","submitted_at":"2025-02-25T17:58:02Z","title":"AgentRM: Enhancing Agent Generalization with Reward Modeling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.18407","snapshot_observed_at":"2026-08-06T18:47:22.078905Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.078905Z"},"links":{"cited_paper":"/paper/2502.18407","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:d1c3264582ac185deff563332773505286b2a1c66367e9d2cb159dc4b0bf77b8","observation_id":"b0bc5883-362f-47e2-b5bd-e51e120a841e","resolution":{"observed_at":"2026-08-06T18:47:22.078905Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03136","last_updated":"2025-08-20T03:44:04Z","snapshot_observed_at":"2026-07-06T19:27:29.609371Z","submitted_at":"2024-10-04T04:23:36Z","title":"Deliberate Reasoning in Language Models as Structure-Aware Planning with an Accurate World Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03136","snapshot_observed_at":"2026-08-06T18:47:22.180797Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.180797Z"},"links":{"cited_paper":"/paper/2410.03136","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:2360115061c054573664ca1b697be0309556a8d7331905aacb59858ab975e2f8","observation_id":"099e01cc-e3d8-45c5-af70-45b54d5a7a0c","resolution":{"observed_at":"2026-08-06T18:47:22.180797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.02682","last_updated":"2025-09-10T16:45:42Z","snapshot_observed_at":"2026-07-06T20:46:37.193819Z","submitted_at":"2025-03-04T14:54:45Z","title":"MPO: Boosting LLM Agents with Meta Plan Optimization","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.02682","snapshot_observed_at":"2026-08-06T18:47:22.268797Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.268797Z"},"links":{"cited_paper":"/paper/2503.02682","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:20f80bf15846dfd158cfea99bbdcbf5a61b5ddaccfedce26eb3b288e0b4aff9d","observation_id":"8a51431e-3524-489a-8c88-44714305f71a","resolution":{"observed_at":"2026-08-06T18:47:22.268797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:22.338609Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.338609Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:e4d6d1780e97a2000f2d86f2a393812930014e65ff2c5b32f41baa40e4fafd6b","observation_id":"192b2273-cd1a-48aa-ac22-ec79c06ddf31","resolution":{"observed_at":"2026-08-06T18:47:22.338609Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-06T18:47:22.420291Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.420291Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:2ef83be1ef87c4e33ec731931637ffd5fd310d6bf35f640d63da323540746db4","observation_id":"eb61d8e1-ed56-40a8-80e1-784696faf346","resolution":{"observed_at":"2026-08-06T18:47:22.420291Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:22.629414Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.629414Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:44357a52fd6834dbac61de3811efe3d8bf9748bb2b5a098445eb1b7c49df64bf","observation_id":"ad3bb6f5-ff79-4a3f-b4b1-70497cacb9c7","resolution":{"observed_at":"2026-08-06T18:47:22.629414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:22.792944Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.792944Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:f9ee6d178267d2850f5acbd9fc390edaa2934b0847b352dd4157414fcf2ffa51","observation_id":"6659a5b5-b057-4067-bf44-5df7e27399e9","resolution":{"observed_at":"2026-08-06T18:47:22.792944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.059954Z","title":null,"venue":null,"work_id":"ba6e3a2f-486b-44b3-9e4a-6be2b68ddf0d","year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:22.943545Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:aa337fcee5e59daad82324d22fbd79679f84554dc51b69e3dc803e7d0eb5966e","observation_id":"bfaa7b3a-f669-4c3d-badf-bb1c07864b51","resolution":{"observed_at":"2026-08-06T18:47:24.063051Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11425","last_updated":"2025-03-24T10:18:56Z","snapshot_observed_at":"2026-07-06T20:23:22.709846Z","submitted_at":"2025-01-20T11:46:04Z","title":"Agent-R: Training Language Model Agents to Reflect via Iterative Self-Training","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.11425","snapshot_observed_at":"2026-08-06T18:47:23.023790Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.023790Z"},"links":{"cited_paper":"/paper/2501.11425","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:6715520434089d2e4764a354a2df19d785cd5365ce2ad8c2c47708d2aeb4e4da","observation_id":"5e784283-0055-4fc5-9ae7-06df5910fdd3","resolution":{"observed_at":"2026-08-06T18:47:23.023790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01825","last_updated":"2023-09-13T03:57:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-03T15:34:01Z","title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.01825","snapshot_observed_at":"2026-08-06T18:47:23.075419Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.075419Z"},"links":{"cited_paper":"/paper/2308.01825","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:d8d840a1e7a3784bb6258f049ec33003c56b5b09ce5b3c9f5ca7e6296cd68084","observation_id":"1bd3cfd6-4049-470e-af4d-e5c63b7094da","resolution":{"observed_at":"2026-08-06T18:47:23.075419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.050262Z","title":null,"venue":null,"work_id":"26b2138e-1e57-473d-b6bb-5990278409bf","year":2022},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.191916Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:6cb12f323637b5c822baad41edf6b5fabb66217ca41caad76595f13d52e88830","observation_id":"e4efc397-8e73-4806-ab5e-4ab405074120","resolution":{"observed_at":"2026-08-06T18:47:24.053599Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:23.310760Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.310760Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:b7f9d9e60cdb6908f0f29118f700a5d0b1f74c03094568e7617f604841a0fdc8","observation_id":"55f2cdf5-e890-4e21-bc6b-14ac32574dba","resolution":{"observed_at":"2026-08-06T18:47:23.310760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:24.012163Z","title":null,"venue":null,"work_id":"e237108a-2e93-42ee-b140-54d64e448955","year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.322550Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:b52bc23f26ca8143f15a5f4fc2bf4c36501ded8096c4c1d515fc5b9382a4cd2f","observation_id":"a3b23f49-3da8-4173-9fa1-fe4e5894c6aa","resolution":{"observed_at":"2026-08-06T18:47:24.041514Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09959","last_updated":"2025-01-17T05:21:49Z","snapshot_observed_at":"2026-07-06T20:22:19.281265Z","submitted_at":"2025-01-17T05:21:49Z","title":"A Survey on Multi-Turn Interaction Capabilities of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09959","snapshot_observed_at":"2026-08-06T18:47:23.325662Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.325662Z"},"links":{"cited_paper":"/paper/2501.09959","citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:1ab405ecee73a4a39a716424b1bb5be3ad8c43c9dbb12dc0e856115ed5caece6","observation_id":"bff288c3-8dd9-4cdd-a884-46bef3babba4","resolution":{"observed_at":"2026-08-06T18:47:23.325662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:47:23.789171Z","title":null,"venue":null,"work_id":"339e5907-04e6-4447-b743-437403372636","year":2025},"citing_paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-06T18:47:23.367773Z"},"links":{"citing_paper":"/paper/2507.07441"},"observation_digest":"sha256:76a75b068d6cfd5e3912ea1d5b0e61c7f3522876df1ae773468941cb014425a2","observation_id":"7da4d6c5-15f8-4e26-8e69-46c479e6fc78","resolution":{"observed_at":"2026-08-06T18:47:23.992294Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.07441","last_updated":"2025-08-20T22:10:48Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T18:38:28.555846Z","submitted_at":"2025-07-10T05:38:15Z","title":"SAND: Boosting LLM Agents with Self-Taught Action Deliberation"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":38,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 0 inbound Pith citation observations for arXiv:2507.07441."}