{"as_of":"2026-08-01T22:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d436076ec61186d7b2dd2f49a1f4ff3451a3f7cdb75ca805e96dfde333c111e8","coverage":[{"denominator":82,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":82,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T05:02:49.206053Z","state":"measured"},{"denominator":83,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":83,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-01T06:32:01.292127+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T07:34:39.055539Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-01T12:16:17.934468Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"cited_work":{"arxiv_id":"2605.12004","doi":"10.48550/arxiv.2605.12004","metadata_source":"pith","pith_arxiv_id":"2605.12004","snapshot_observed_at":"2026-08-01T12:16:17.934468Z","title":"Learning Agentic Policy from Action Guidance","venue":"cs.CL","work_id":"cd596862-59c6-4c4d-99ac-c6ed278a2dba","year":2026},"citing_paper":{"arxiv_id":"2607.21419","last_updated":"2026-07-30T03:55:54Z","snapshot_observed_at":"2026-08-01T22:13:59.834649Z","submitted_at":"2026-07-23T15:24:35Z","title":"PATS: Policy-Aware Training Scaffolding for Agentic Reinforcement Learning","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-01T07:34:39.055539Z"},"links":{"cited_paper":"/paper/2605.12004","citing_paper":"/paper/2607.21419"},"observation_digest":"sha256:9f5ff5f529cc39c18f46a77909dfcb017035635da2ee4f452af794132ae327f4","observation_id":"1e1e43b4-438e-4481-bbd8-f32e36544441","resolution":{"observed_at":"2026-08-01T07:39:17.666069Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T10:08:10.546089+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.546089+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.546089+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.12004/citation-record","integrity":"/paper/2605.12004/integrity","json":"/paper/2605.12004/citation-record.json","paper":"/paper/2605.12004"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Claude Opus 4.6 model card","venue":null,"work_id":"3bb72bcf-3887-42cb-9f78-ca4e454cad7f","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:ad67c3ab7aa8ea15e2650db7183fc667a63a60e3e0f0d89c77df01f5441f219e","observation_id":"1639d43e-7ea4-4e77-9b48-1f80e58be624","resolution":{"observed_at":"2026-05-13T11:07:39.831018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.07982","last_updated":"2025-06-09T17:52:18Z","snapshot_observed_at":"2026-07-06T21:39:13.304260Z","submitted_at":"2025-06-09T17:52:18Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","version":1},"cited_work":{"arxiv_id":"2506.07982","doi":"10.48550/arxiv.2506.07982","metadata_source":"pith","pith_arxiv_id":"2506.07982","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"$\\tau^2$-Bench: Evaluating Conversational Agents in a Dual-Control Environment","venue":"cs.AI","work_id":"3a498b1a-455f-4667-b572-c5216c99a89c","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.07982","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:82125d4d0d943f025f812ea70bd7c57cc7de26514797dac274cfcc25578ff662","observation_id":"5691e81c-db21-469e-914f-a15cb41526d4","resolution":{"observed_at":"2026-05-13T05:07:17.802510Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:22.129969+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fine- tuning web agents: It works, but it’s trickier than you think","venue":null,"work_id":"27825400-9f46-4f37-a856-c78af8d6001d","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:5eb03f813598a01a5be8e359b9661ba759708428e3d1eb732294675772465bb5","observation_id":"cf10f969-fe6e-4803-be0b-9cd962e0eda4","resolution":{"observed_at":"2026-05-13T11:07:39.816125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11468","last_updated":"2025-04-10T16:54:05Z","snapshot_observed_at":"2026-07-06T21:09:55.394184Z","submitted_at":"2025-04-10T16:54:05Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2504.11468","doi":"10.48550/arxiv.2504.11468","metadata_source":"pith","pith_arxiv_id":"2504.11468","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","venue":"cs.CL","work_id":"a521360c-8673-4d0d-a3a3-6eb9f7a71b90","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.11468","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:25c6da17003630b187b4428990806d2ed9a2e39a734d5d6fb1313dd5d07d1af1","observation_id":"f2176436-10eb-4ea0-bbf8-da2b115a9287","resolution":{"observed_at":"2026-05-17T15:43:34.765363Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13651","last_updated":"2025-06-16T16:16:14Z","snapshot_observed_at":"2026-07-06T21:43:03.415527Z","submitted_at":"2025-06-16T16:16:14Z","title":"xbench: Tracking Agents Productivity Scaling with Profession-Aligned Real-World Evaluations","version":1},"cited_work":{"arxiv_id":"2506.13651","doi":"10.48550/arxiv.2506.13651","metadata_source":"pith","pith_arxiv_id":"2506.13651","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"xbench: Tracking agents productivity scaling with profession-aligned real-world evaluations","venue":"cs.LG","work_id":"b7f0bdf6-3821-4735-b55e-be58f6ef326d","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.13651","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f34329f63793d15e21f9fabdca990bc3328046f907c54f5a5c49c62e52d3e65c","observation_id":"a2d792c1-5c8e-444f-8ae2-27ecb873b73a","resolution":{"observed_at":"2026-05-13T05:07:17.799791Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.06948","last_updated":"2026-05-30T09:01:09Z","snapshot_observed_at":"2026-07-06T22:26:03.668339Z","submitted_at":"2025-09-08T17:58:02Z","title":"Beyond Two-Stage Training: Cooperative SFT and RL for LLM Reasoning","version":3},"cited_work":{"arxiv_id":"2509.06948","doi":"10.48550/arxiv.2509.06948","metadata_source":"pith","pith_arxiv_id":"2509.06948","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Beyond two-stage training: Cooperative sft and rl for llm reasoning","venue":"cs.CL","work_id":"c187c8ff-10ae-4b69-9b4d-8a04c661f929","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2509.06948","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1e17f7878945f85b75bcd91600da6c3e478a15c24c67993e263401f1ad8dbf38","observation_id":"357d6141-dc4b-4d87-8a14-a127566deef6","resolution":{"observed_at":"2026-06-02T02:03:33.109892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T09:34:48.462851Z","title":"GPG: A simple and strong reinforcement learning baseline for model reasoning","venue":null,"work_id":"97c1efeb-7277-42d6-b9c1-53580a47bd15","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:8574787c317ca63d8b3c9a482242644574b53b7eff9c827e7740dca32df3b0dd","observation_id":"987e40b8-e036-4fc2-85eb-758838d08a28","resolution":{"observed_at":"2026-05-13T11:07:39.825455Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.14234","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T20:05:34.003466Z","title":"Redsearcher: A scalable and cost-efficient framework for long-horizon search agents","venue":null,"work_id":"fbfb7693-53da-4498-8b80-3e1c3d3dd9b1","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:4cd46c7774e82200b08a90642d01b5799d5c702b35b5f63c48f57f39a135e886","observation_id":"fbe351f8-dc2b-4f91-8b4d-b0e3a4584bae","resolution":{"observed_at":"2026-05-13T05:07:17.587389Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Harder is better: Boosting mathematical reasoning via difficulty-aware GRPO and multi-aspect question reformulation","venue":null,"work_id":"9a02ba42-e3b3-4e6a-a3d6-302a3d1440c8","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b614b9eadc73672098af7044b79009524517e90b12648af97ad547d2c74c6387","observation_id":"b3e77c83-4a51-4118-9c6d-fc14d1fed60d","resolution":{"observed_at":"2026-05-13T11:07:39.776771Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T16:36:22.361197Z","title":"Mind2web: Towards a generalist agent for the web.Advances in Neural Information Processing Systems, 36:28091–28114","venue":null,"work_id":"11ebf5e9-ce0a-48b6-8ddf-e267c900004c","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:00e4c31d21a60d01084771b9317a5b9737bf0ddeeb47b23e4c6d81f1bc2b3b70","observation_id":"de9932e6-8f5c-4a75-86a9-c4c8d28d60d8","resolution":{"observed_at":"2026-05-13T11:07:39.818141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17352","last_updated":"2025-11-11T08:13:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-21T17:52:43Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","version":3},"cited_work":{"arxiv_id":"2503.17352","doi":"10.48550/arxiv.2503.17352","metadata_source":"pith","pith_arxiv_id":"2503.17352","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","venue":"cs.CV","work_id":"de4c64e1-82b1-4f70-8311-a3539e7bf400","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.17352","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f6d71cd47a9fb7580519f4b8b8c6612e445b957c8986711a41985080b6f9001d","observation_id":"f5550734-41ba-41e7-80b5-8341adc77647","resolution":{"observed_at":"2026-05-19T06:59:03.519833Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wildclawbench","venue":null,"work_id":"367116cb-6af4-4059-ac0a-34cfb7ac6999","year":null},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:14c7d7feffa1beabd6b530fadc4475041138b5b5239a6f7a177885f18418f348","observation_id":"a0d8fdd6-9f20-4899-a0f7-7fd7a4f750e6","resolution":{"observed_at":"2026-05-13T11:07:39.823623Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.16410","last_updated":"2025-05-22T09:00:19Z","snapshot_observed_at":"2026-07-06T21:28:24.644692Z","submitted_at":"2025-05-22T09:00:19Z","title":"Tool-Star: Empowering LLM-Brained Multi-Tool Reasoner via Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2505.16410","doi":"10.48550/arxiv.2505.16410","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.16410","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Tool-star: Empowering LLM-brained multi-tool reasoner via reinforcement learning","venue":null,"work_id":"4b5aa09c-2d1c-4591-888d-3e9aa8a1a0dc","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2505.16410","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f9d07c19bebb87c4f3cfe54987450933b1f262da39e842b82493c6c426c1a7de","observation_id":"751d4fcb-e620-4ec5-abcc-dc6947058c96","resolution":{"observed_at":"2026-05-13T05:07:17.727198Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.19849","last_updated":"2025-07-26T07:53:11Z","snapshot_observed_at":"2026-07-06T22:03:15.296567Z","submitted_at":"2025-07-26T07:53:11Z","title":"Agentic Reinforced Policy Optimization","version":1},"cited_work":{"arxiv_id":"2507.19849","doi":"10.48550/arxiv.2507.19849","metadata_source":"pith","pith_arxiv_id":"2507.19849","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Agentic Reinforced Policy Optimization","venue":"cs.LG","work_id":"6cb0d241-5e4d-44ed-9476-659819ec0681","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2507.19849","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6c5eb0b066eb9e68074cbf7fa7216179c7332e7c80b5db72dd279220dd3b9280","observation_id":"37eda508-fa87-42b7-be1d-f85b7e9c116e","resolution":{"observed_at":"2026-05-17T02:57:12.173630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.18292","last_updated":"2026-04-20T14:01:10Z","snapshot_observed_at":"2026-07-06T23:05:13.178333Z","submitted_at":"2026-04-20T14:01:10Z","title":"Agent-World: Scaling Real-World Environment Synthesis for Evolving General Agent Intelligence","version":1},"cited_work":{"arxiv_id":"2604.18292","doi":"10.48550/arxiv.2604.18292","metadata_source":"pith","pith_arxiv_id":"2604.18292","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Agent-World: Scaling Real-World Environment Synthesis for Evolving General Agent Intelligence","venue":"cs.AI","work_id":"ffadde38-12ae-49ed-b435-0d399a0ea2fd","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2604.18292","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1cf0e49c5afc8c39dfcc9bd74b0192c7805650ca5acbb1b045b0795d445f8349","observation_id":"bec250b6-7d34-4c30-bf01-d732a8d11b3a","resolution":{"observed_at":"2026-05-13T05:07:17.621909Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09572","last_updated":"2025-04-22T17:56:22Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T17:40:52Z","title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","version":3},"cited_work":{"arxiv_id":"2503.09572","doi":"10.48550/arxiv.2503.09572","metadata_source":"pith","pith_arxiv_id":"2503.09572","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Plan-and-Act: Improving Planning of Agents for Long-Horizon Tasks","venue":"cs.CL","work_id":"ee84a785-1e36-4af7-b0e8-13fc86cba1ea","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.09572","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:021c6cbdeb939a4584adb8fb9688add004bc4277afdccef59f9ebd620a1f7222","observation_id":"5efedcd7-945a-4fa4-8eda-6e88d8374c97","resolution":{"observed_at":"2026-05-17T21:32:18.834213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.10978","last_updated":"2025-10-28T15:11:36Z","snapshot_observed_at":"2026-07-29T19:20:21.974239Z","submitted_at":"2025-05-16T08:26:59Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","version":3},"cited_work":{"arxiv_id":"2505.10978","doi":"10.48550/arxiv.2505.10978","metadata_source":"pith","pith_arxiv_id":"2505.10978","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Group-in-Group Policy Optimization for LLM Agent Training","venue":"cs.LG","work_id":"bc65d492-e6ba-4522-874c-43d2f4fc5191","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2505.10978","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:67be521452a36299780cc5336db272f3406e4685b3fe5e70a95b60a44e539ed5","observation_id":"4a61f7ee-90f2-460f-9b23-03f66a981999","resolution":{"observed_at":"2026-05-13T05:07:17.632913Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.19767","last_updated":"2025-06-24T16:31:37Z","snapshot_observed_at":"2026-07-06T21:47:01.972372Z","submitted_at":"2025-06-24T16:31:37Z","title":"SRFT: A Single-Stage Method with Supervised and Reinforcement Fine-Tuning for Reasoning","version":1},"cited_work":{"arxiv_id":"2506.19767","doi":"10.48550/arxiv.2506.19767","metadata_source":"arxiv_reference","pith_arxiv_id":"2506.19767","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2506.19767 , year=","venue":null,"work_id":"03d392b0-d6dc-44a8-bfdd-888ccbc9e68e","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.19767","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9d182d7830be3c38c3ceb71d97621dda3580078affc26b3009e9a3fb0b108d15","observation_id":"859ffaa4-7137-4c9d-bc2d-2b543d12052e","resolution":{"observed_at":"2026-05-13T05:07:17.645623Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.07976","doi":"10.48550/arxiv.2508.07976","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T17:27:26.699065Z","title":"Beyond ten turns: Unlocking long-horizon agentic search with large-scale asynchronous rl","venue":null,"work_id":"cf9b5d87-269e-421e-aec4-52f7899c89f0","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:5b00d30199d338e9f3bb663383049ff5a3aaf95124a5e67efb13971d104d4716","observation_id":"1e817b8d-464e-40b4-b648-316d0bf06b9a","resolution":{"observed_at":"2026-05-13T05:07:17.649435Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.20532","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Actor-curator: Co-adaptive curriculum learning via policy-improvement bandits for rl post-training.arXiv preprint arXiv:2602.20532","venue":null,"work_id":"a8bc50a7-b768-4a64-a334-1411302abeb2","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:341f2c5ebb10edd0675ea737ae97c61b13a90f51f733fbf220b735ae8c2e9975","observation_id":"009f5b83-d135-430d-a75b-484f7a33c050","resolution":{"observed_at":"2026-05-13T05:07:17.590870Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deep q-learning from demonstrations","venue":null,"work_id":"58a37b8d-dae6-41eb-87bb-f614f0a8706d","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:3b56c59513207bd00c4c7be47bac75b6298e3fbd7766675e1bbd787602fa8234","observation_id":"3af81f34-30cf-47d1-bbd8-d1e391518386","resolution":{"observed_at":"2026-05-13T11:07:39.778716Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Boosting mllm reasoning with text-debiased hint-grpo","venue":null,"work_id":"fb653f7a-bc25-4aad-969d-efc8dbdea9e7","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:8e6213e0ed5b4ca0292d7b49cae2c9a1d5a8490fa79bc75b8e321f21ce07de42","observation_id":"50fbd06d-8e5a-424c-bb98-d95aa186f427","resolution":{"observed_at":"2026-05-13T11:07:39.798467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.20802","last_updated":"2026-02-16T14:49:34Z","snapshot_observed_at":"2026-07-29T19:52:32.104228Z","submitted_at":"2026-01-28T17:45:12Z","title":"Reinforcement Learning via Self-Distillation","version":2},"cited_work":{"arxiv_id":"2601.20802","doi":"10.48550/arxiv.2601.20802","metadata_source":"pith","pith_arxiv_id":"2601.20802","snapshot_observed_at":"2026-07-10T16:57:24.584548Z","title":"Reinforcement Learning via Self-Distillation","venue":"cs.LG","work_id":"b193541d-5853-4ea4-8e4b-8e4c08617eb6","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.20802","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:239d5bf2850f3cd306a9e42c4d3e8345cf0b46313c6ba5c105a84bd32cd5bdd1","observation_id":"b82336af-dcd4-47ec-bc91-ffc20befc04b","resolution":{"observed_at":"2026-05-13T05:07:17.597502Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-05-21T06:23:12.649775+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T06:23:12.649775+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tree search for LLM agent reinforcement learning","venue":null,"work_id":"a6cdb788-4cab-4552-972f-25ecc0d4d0f0","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:2b48b565f25a7ce809398e5b2728f32522d6231436b25b2c8c7234434e4e8864","observation_id":"08144029-b394-4e13-8a6d-176e06855d6c","resolution":{"observed_at":"2026-05-13T11:07:39.780576Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Thinking with map: Reinforced parallel map-augmented agent for geolocalization.ACL","venue":null,"work_id":"66f524d3-3245-4876-88c5-99db2e3be337","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:c6cc4adfea0c06bc660010cb08e632991fdc9b261bb8d06cb989633d3245f660","observation_id":"0be9105e-bea4-4158-beac-d4f44700b3c5","resolution":{"observed_at":"2026-05-13T11:07:39.812409Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.19803","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T18:30:01.575162Z","title":"Vcrl: Variance-based curriculum reinforcement learning for large language models","venue":null,"work_id":"87da4de6-b94c-43ee-8a28-bd43363bea62","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:3da4d045087a0d4eb738d4c4d062d9694043e39ead3fd78c11b38261a1e8baa3","observation_id":"8635f681-43ae-46dd-94c9-dfaef4cb5faa","resolution":{"observed_at":"2026-05-13T05:07:17.676707Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":"2310.06770","doi":"10.1145/512927.512945","metadata_source":"pith","pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","venue":"cs.CL","work_id":"d0effe15-a689-441a-8e3f-ea35f1c4e4b1","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d07ffadd434051207383ede43f9af03e0102790cc8a410910083a8b3f73bfd70","observation_id":"80061224-480d-4b83-b19b-6846e08e04c9","resolution":{"observed_at":"2026-05-13T05:07:17.688917Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.09516","last_updated":"2025-08-05T19:08:38Z","snapshot_observed_at":"2026-07-06T20:51:28.022519Z","submitted_at":"2025-03-12T16:26:39Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","version":5},"cited_work":{"arxiv_id":"2503.09516","doi":"10.48550/arxiv.2503.09516","metadata_source":"pith","pith_arxiv_id":"2503.09516","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning","venue":"cs.CL","work_id":"0e0b7549-2bc4-4574-aa7f-588ffa16eaae","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.09516","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:ee8d7f2cbd60952fa184d17d20943cd8b407ebb4c2cfa60909b1601259831425","observation_id":"afda1e21-8088-47ae-9ae2-b7a6effdc9d4","resolution":{"observed_at":"2026-05-13T05:07:17.606250Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02592","last_updated":"2025-07-03T12:59:07Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-03T12:59:07Z","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","version":1},"cited_work":{"arxiv_id":"2507.02592","doi":"10.48550/arxiv.2507.02592","metadata_source":"pith","pith_arxiv_id":"2507.02592","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","venue":"cs.CL","work_id":"fec5a195-1dd5-425a-b552-109af948a7dd","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2507.02592","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f1f40b51321e4bec79f6c2190efcda1852841800ff3853828b4b9787ef52f1f8","observation_id":"52a638ee-6fb7-4688-97ab-09df64d5fd0f","resolution":{"observed_at":"2026-05-17T15:37:09.773663Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Adacurl: Adaptive curriculum reinforcement learning with invalid sample mitigation and historical revisiting","venue":null,"work_id":"7da0c753-7974-4e7c-8723-e7d29f37508c","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:de9017bdc6d6e350098aee339ab88d097acfb430080f8f773427753616095f24","observation_id":"b970073b-9c08-4745-9f20-11aae7c9e7b9","resolution":{"observed_at":"2026-05-13T11:07:39.814168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.21776","last_updated":"2025-10-13T12:40:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-30T16:25:25Z","title":"WebThinker: Empowering Large Reasoning Models with Deep Research Capability","version":2},"cited_work":{"arxiv_id":"2504.21776","doi":"10.48550/arxiv.2504.21776","metadata_source":"pith","pith_arxiv_id":"2504.21776","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"WebThinker: Empowering Large Reasoning Models with Deep Research Capability","venue":"cs.CL","work_id":"7e319d34-eb88-4ea9-8c0d-b66320599f98","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.21776","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:8a28818885d634cc87ee744d688765f31b080e399328ef1d68fbf2b120d74378","observation_id":"55358957-318a-4682-941b-47aa643d0dac","resolution":{"observed_at":"2026-05-16T19:14:25.573630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06892","last_updated":"2025-07-11T10:32:34Z","snapshot_observed_at":"2026-07-06T21:54:29.624793Z","submitted_at":"2025-07-09T14:29:45Z","title":"Squeeze the Soaked Sponge: Efficient Off-policy Reinforcement Finetuning for Large Language Model","version":3},"cited_work":{"arxiv_id":"2507.06892","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.06892","snapshot_observed_at":"2026-07-04T08:09:40.703268Z","title":"Squeeze the soaked sponge: Efficient off-policy reinforcement finetuning for large language model","venue":null,"work_id":"3acc1af6-eff2-4ff7-9eea-ab00848acbb3","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2507.06892","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f0ad86183e16d26451eb24c99f7537ea3322477dc84dd58ab925579a0c1e353f","observation_id":"92c5b202-9bfe-4c5c-b64a-08617c072a60","resolution":{"observed_at":"2026-05-13T05:07:17.618672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Guided exploration with proximal policy optimization using a single demonstration","venue":null,"work_id":"e347e5ff-6f83-4823-b329-6efb3c5c3bc0","year":2021},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:17288c7920b55e3439e8892c13e6c52153b020e4c15c56268d484d592332f2d1","observation_id":"cfd20f2a-199e-4d14-ae46-5d9da119b7bb","resolution":{"observed_at":"2026-05-13T11:07:39.821791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T03:54:30.352907Z","title":"Truthfulqa: Measuring how models mimic hu- man falsehoods","venue":null,"work_id":"81229962-3656-44da-a54c-197faeefdb7f","year":2022},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:01846c8c14b34b1205ff92a71ec5d0cdfb6e6faf72afd6838711771010815810","observation_id":"66b3e568-7ed0-4777-bb46-04d01342c9e7","resolution":{"observed_at":"2026-05-13T11:07:39.820085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.19437","doi":"10.1016/j.neucom.2023.127063.url:https://www.sciencedirect","metadata_source":"pith","pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeek-V3 Technical Report","venue":"cs.CL","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6e7bde249ade78b5bed1ec9ce04708e2e9155ff2d3b28646716838dd243eb7ab","observation_id":"8caa542f-b769-4914-8565-be8a51ce2d1a","resolution":{"observed_at":"2026-05-13T05:07:17.796733Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21460","last_updated":"2025-03-27T12:50:17Z","snapshot_observed_at":"2026-07-06T20:59:35.694800Z","submitted_at":"2025-03-27T12:50:17Z","title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","version":1},"cited_work":{"arxiv_id":"2503.21460","doi":"10.1145/3573051.3596191","metadata_source":"pith","pith_arxiv_id":"2503.21460","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","venue":"cs.CL","work_id":"4aff48d9-c46d-41d9-904b-52251e559596","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.21460","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:c9983fcc9973f1aa2e7a99b173b9cbc6904d16bc44ceeed0d14d413e9b588dca","observation_id":"2361523b-ecd7-42c7-a7aa-dac6d962aa89","resolution":{"observed_at":"2026-05-13T05:07:17.594178Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.07527","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:25:57.811550Z","title":"Learning what reinforcement learning can't: Interleaved online fine-tuning for hardest questions","venue":null,"work_id":"77400081-d7b1-4be9-88ff-c1082c417c04","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d122df9e3d9212e6c54ee734285df4464b603acce4dcc67f7ff43a4b4023798c","observation_id":"b737ab17-d520-460c-a169-a0d582078b9c","resolution":{"observed_at":"2026-05-13T05:07:17.787952Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.08377","last_updated":"2026-04-09T15:38:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-09T15:38:27Z","title":"SkillClaw: Let Skills Evolve Collectively with Agentic Evolver","version":1},"cited_work":{"arxiv_id":"2604.08377","doi":"10.48550/arxiv.2604.08377","metadata_source":"pith","pith_arxiv_id":"2604.08377","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"SkillClaw: Let Skills Evolve Collectively with Agentic Evolver","venue":"cs.AI","work_id":"31a44ebb-9aae-4719-8d62-50f8c82ece16","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2604.08377","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:cb844cf8bfc7cf269376b1fe21ef52523bbdddd1d3dbec7eb06be3266d4ae0c9","observation_id":"826ac4a2-1c2d-4ade-9115-ee3388b75676","resolution":{"observed_at":"2026-05-13T05:07:17.776747Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":"2311.12983","doi":"10.48550/arxiv.2311.12983","metadata_source":"pith","pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"GAIA: a benchmark for General AI Assistants","venue":"cs.CL","work_id":"cf222b33-f7a3-4044-a570-ecfe25edb3f8","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:2e7c784caff423c39721e8539919c2ba3229c06ecd416385c652d16f484fab35","observation_id":"cf36beac-9f5d-45f3-9e96-e84907002074","resolution":{"observed_at":"2026-05-13T05:07:17.603371Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minimax m2.1 system card","venue":null,"work_id":"730e5b8d-7000-488c-9372-760f82058769","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:8bc24847ea2557f81e441b7f0fe74e29ae2b60cb58e89ed12ce268b1aa75b7d7","observation_id":"14e1b130-4e90-4896-aea3-d3393945bed6","resolution":{"observed_at":"2026-05-13T11:07:39.829342Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Over- coming exploration in reinforcement learning with demonstrations","venue":null,"work_id":"e26d58db-8f2b-4b6c-8b96-41183f4b6a4c","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:94fad949333038804ffe16998ff8d4b104dd52f4f0f64e901c4128df303894ab","observation_id":"bdfa9d7a-d43d-4f0b-b3dc-39f5de834860","resolution":{"observed_at":"2026-05-13T11:07:39.806867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13923","last_updated":"2025-06-20T00:51:15Z","snapshot_observed_at":"2026-07-06T21:43:12.752792Z","submitted_at":"2025-06-16T19:03:06Z","title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models","version":2},"cited_work":{"arxiv_id":"2506.13923","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.13923","snapshot_observed_at":"2026-07-03T20:48:56.210532Z","title":"Shubham Parashar, Shurui Gui, Xiner Li, Hongyi Ling, Sushil Vemuri, Blake Olson, Eric Li, Yu Zhang, James Caverlee, Dileep Kalathil, and Shuiwang Ji","venue":null,"work_id":"a0eba3eb-bd91-4d3d-bb05-c395ae033c2d","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2506.13923","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b786d9d5499c9b9201d69b844b558e5a2b72262526b920c6fc3afb4c059c3fcd","observation_id":"2cf22fff-26b7-4055-817f-97f6907687bc","resolution":{"observed_at":"2026-05-13T05:07:17.751601Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-5.4 thinking system card","venue":null,"work_id":"f4cbca14-81cb-4c76-8915-d299cc615ba7","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:cd054e35205bd382f83d7b70bb9697320064cae4a3332780c4c2fcda311aa64d","observation_id":"a00052b6-468d-4553-a72a-21ff3ba94931","resolution":{"observed_at":"2026-05-13T11:07:39.810613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Iterative reasoning preference optimization.Advances in Neural Information Processing Systems, 37:116617–116637","venue":null,"work_id":"170354cc-06ed-403b-995e-f5c2c734d76a","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6d2e557ab9d65e0c78b98755fafc916cb73dfeb6e781a4886cb1c71f87c9ff55","observation_id":"b097e1fa-7b21-481a-acf6-bde3b98e240a","resolution":{"observed_at":"2026-05-13T11:07:39.802870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12326","last_updated":"2025-01-21T17:48:10Z","snapshot_observed_at":"2026-07-06T20:23:58.426780Z","submitted_at":"2025-01-21T17:48:10Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","version":1},"cited_work":{"arxiv_id":"2501.12326","doi":"10.48550/arxiv.2501.12326","metadata_source":"pith","pith_arxiv_id":"2501.12326","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"UI-TARS: Pioneering Automated GUI Interaction with Native Agents","venue":"cs.AI","work_id":"0bbcf263-a46d-4525-a438-11fce3316568","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2501.12326","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:2aadb017a4bb38702c1761152d55842cf5fa22db1a4380a1a0f081b84373a87c","observation_id":"2fba3040-3165-4db1-8b1a-54dc1ca17d4e","resolution":{"observed_at":"2026-05-13T05:07:17.699644Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:19.638951+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:19.638951+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1709.10087","last_updated":"2018-06-26T13:31:37Z","snapshot_observed_at":"2026-07-06T06:01:51.192687Z","submitted_at":"2017-09-28T17:51:13Z","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","version":2},"cited_work":{"arxiv_id":"1709.10087","doi":"10.48550/arxiv.1709.10087","metadata_source":"pith","pith_arxiv_id":"1709.10087","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","venue":"cs.LG","work_id":"e0799bae-989e-4c06-891d-c93b0b32024d","year":2017},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1709.10087","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1e93a3d6c750f845998de688bd676abcd46b16c2f1ed233f994c1d8b89086a75","observation_id":"a57effd7-b998-4217-8826-62f5937f4fe7","resolution":{"observed_at":"2026-05-13T05:07:17.755910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":"2311.12022","doi":"10.48550/arxiv.2311.12022","metadata_source":"pith","pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-07-10T14:47:14.590391Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","venue":"cs.AI","work_id":"9e2a976b-f5ad-4aee-af5c-243fe0fe75d2","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:fefc1713a935bb9614e7460ac489d6656434569ff99603995ed89d0ea4808bb1","observation_id":"9c44e73f-ff6c-4474-9312-9bd559a419c2","resolution":{"observed_at":"2026-05-13T05:07:17.773989Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T04:38:46.941438+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T04:38:46.941438+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.06347","last_updated":"2017-08-28T09:20:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-07-20T02:32:33Z","title":"Proximal Policy Optimization Algorithms","version":2},"cited_work":{"arxiv_id":"1707.06347","doi":"10.1016/j.artint.2010.12.005","metadata_source":"pith","pith_arxiv_id":"1707.06347","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Proximal Policy Optimization Algorithms","venue":"cs.LG","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","year":2017},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1707.06347","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:3345ae642ce4db18a75ea5c7f8fb4613243fae2e04e9cc9cbc39f632886d8202","observation_id":"d467159e-4082-404f-8d31-199b819351da","resolution":{"observed_at":"2026-05-13T05:07:17.600157Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:5dbc07dd4c1f41654f6a4cf121d9c26df28f685f5a9595f2eae4de9529cc626c","observation_id":"1633b736-430c-4d9c-862d-6bc615c3575a","resolution":{"observed_at":"2026-05-13T05:07:17.696982Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.19897","last_updated":"2026-01-27T18:59:08Z","snapshot_observed_at":"2026-07-06T22:43:12.468415Z","submitted_at":"2026-01-27T18:59:08Z","title":"Self-Distillation Enables Continual Learning","version":1},"cited_work":{"arxiv_id":"2601.19897","doi":"10.48550/arxiv.2601.19897","metadata_source":"pith","pith_arxiv_id":"2601.19897","snapshot_observed_at":"2026-07-10T16:57:24.571259Z","title":"Self-Distillation Enables Continual Learning","venue":"cs.LG","work_id":"e9aa25e3-870c-46c8-8270-e4e5948d09f0","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.19897","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:c503c5789c9ae9ce3c25277279687964077a85ca5aae80899d1a467ab37b2f8d","observation_id":"fc9c0f91-b266-4d95-b705-4562cdd3e576","resolution":{"observed_at":"2026-05-13T05:07:17.685825Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-05-23T10:52:50.091537+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T10:52:50.091537+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-07-06T22:40:51.136169Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":"2601.03267","doi":"10.48550/arxiv.2601.03267","metadata_source":"pith","pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-07-11T02:17:46.480513Z","title":"OpenAI GPT-5 System Card","venue":"cs.CL","work_id":"ca87689a-0d29-4476-b504-b65dbbb08af4","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:37223991bc4bb5576c74f581fb98a32fc66de4c756fee118bc952572178dfd91","observation_id":"fa15cef6-ae82-438f-9a32-97dd9aa78c15","resolution":{"observed_at":"2026-05-13T05:07:17.679643Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-07-11T05:19:25.151918+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T05:19:25.151918+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":"2602.02276","doi":"10.48550/arxiv.2602.02276","metadata_source":"pith","pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Kimi K2.5: Visual Agentic Intelligence","venue":"cs.CL","work_id":"d690be8f-5d53-49b0-b1e7-79668eb8fcdb","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:9085a7c14b909f51adbe58eeda0ea0a02488d16dec2f94533dea19ee7cd03367","observation_id":"95c47958-1b89-4fc5-ad6c-b886c217c52a","resolution":{"observed_at":"2026-05-13T05:07:17.768488Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:04:52.365581Z","title":"Qwen3.5: Accelerating productivity with native multimodal agents, February","venue":null,"work_id":"f37dcb7f-3329-4b47-b2b7-85458b1592d1","year":null},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:62ed918fa7a384e20f10ae5c6276001f536446188bff3325d74a2eaa933f07d3","observation_id":"485864f6-187f-404a-a5ab-3f6c1c9529ab","resolution":{"observed_at":"2026-05-13T11:07:39.805037Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T10:04:52.368885Z","title":null,"venue":null,"work_id":"c39e4053-229e-4a4c-8726-7c6ff6b49736","year":null},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6dee3774e9f42a7d9159b23415b15198cab66cd334ae15e119547b857cbd8006","observation_id":"e2912aa8-4b67-4e68-b97c-f70d1e74a105","resolution":{"observed_at":"2026-05-13T11:07:39.800591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.24701","last_updated":"2026-05-18T04:10:32Z","snapshot_observed_at":"2026-07-06T22:34:18.297603Z","submitted_at":"2025-10-28T17:53:02Z","title":"Tongyi DeepResearch Technical Report","version":3},"cited_work":{"arxiv_id":"2510.24701","doi":"10.48550/arxiv.2510.24701","metadata_source":"pith","pith_arxiv_id":"2510.24701","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Tongyi DeepResearch Technical Report","venue":"cs.CL","work_id":"1c8db01b-b50f-4711-b181-a04bcc3e9aa8","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2510.24701","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:b6d14371b850424a3fdde213ce2c1f57754e99b24018d1a3466f9bce7187c3ad","observation_id":"68746824-b279-4f01-862c-ff4041c7e0d0","resolution":{"observed_at":"2026-05-15T08:56:57.433092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.52202/075280-3275","metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T17:35:10.028423Z","title":"Language models don’t always say what they think: Unfaithful explanations in chain-of-thought prompting.Advances in Neural Information Processing Systems, 36:74952–74965","venue":"Advances in Neural Information Processing Systems 36","work_id":"20491003-4ee3-4877-8191-ba12773d0f0b","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:55c830ac920d501b8378bb2f7ef84c280a004098db139b9affe0f7fbaab7a465","observation_id":"f4e2c5f8-93f7-4bc0-b219-a26eaf80f776","resolution":{"observed_at":"2026-05-13T11:07:39.774687Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T07:38:14.455455+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T07:38:14.455455+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T07:38:14.455455+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1812.02648","last_updated":"2018-12-06T16:36:20Z","snapshot_observed_at":"2026-07-06T07:19:38.028327Z","submitted_at":"2018-12-06T16:36:20Z","title":"Deep Reinforcement Learning and the Deadly Triad","version":1},"cited_work":{"arxiv_id":"1812.02648","doi":"10.48550/arxiv.1812.02648","metadata_source":"pith","pith_arxiv_id":"1812.02648","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Deep Reinforcement Learning and the Deadly Triad","venue":"cs.AI","work_id":"de214ead-4cb0-4abd-be3d-ae3389f55e9b","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1812.02648","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:4c4272001eca6781c803ba6f6640d28eb79e50a442e1df388c097d5e7c7eadf2","observation_id":"8dc39b33-8e3c-4204-ba86-c86d6618adae","resolution":{"observed_at":"2026-05-13T05:07:17.711777Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1707.08817","last_updated":"2018-10-08T13:38:52Z","snapshot_observed_at":"2026-07-06T05:52:59.445968Z","submitted_at":"2017-07-27T11:16:53Z","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","version":2},"cited_work":{"arxiv_id":"1707.08817","doi":null,"metadata_source":"pith","pith_arxiv_id":"1707.08817","snapshot_observed_at":"2026-07-10T21:47:36.312721Z","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","venue":"cs.AI","work_id":"a4478529-fa6c-4c19-98bd-743c55919fd6","year":2017},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/1707.08817","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:4005a7978a300d8ba5b8e9408ea4e84801e4db42a684f68bacba929d74e7db20","observation_id":"87baeeb7-af4a-4cee-b02e-f8467ffb4008","resolution":{"observed_at":"2026-05-13T05:07:17.758867Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.24873","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T21:26:14.207661Z","title":"Let it flow: Agentic crafting on rock and roll, building the rome model within an open agentic learning ecosystem","venue":null,"work_id":"a1afde43-96e3-49c9-af14-25f128d65fe3","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:ddf2c47a0428caf00529278ea77d8ae2864b51f58018e36cbc298adc974e08b1","observation_id":"74e5c648-4a6d-4977-9805-bacabdda8f9f","resolution":{"observed_at":"2026-05-13T05:07:17.761647Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.10165","last_updated":"2026-05-11T10:03:41Z","snapshot_observed_at":"2026-07-06T22:48:35.345421Z","submitted_at":"2026-03-10T18:59:01Z","title":"OpenClaw-RL: Train Any Agent Simply by Talking","version":2},"cited_work":{"arxiv_id":"2603.10165","doi":"10.48550/arxiv.2603.10165","metadata_source":"pith","pith_arxiv_id":"2603.10165","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"OpenClaw-RL: Train Any Agent Simply by Talking","venue":"cs.CL","work_id":"78607317-8305-4515-8dc3-20b4ff5b8f3a","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2603.10165","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:f33e7c7ef47a959e35de4f1782414970f0806a1e4c2e5e8e4a3b08ce9a177eeb","observation_id":"db34b9fc-5d71-432d-993e-62773e8d8907","resolution":{"observed_at":"2026-05-13T05:07:17.765019Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.20073","last_updated":"2025-05-26T17:19:30Z","snapshot_observed_at":"2026-07-06T21:15:59.063396Z","submitted_at":"2025-04-24T17:57:08Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2504.20073","doi":"10.18653/v1/2025.acl-long.887","metadata_source":"pith","pith_arxiv_id":"2504.20073","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","venue":"cs.LG","work_id":"b96383ee-f8dc-471f-aba4-bc5ce9b0b632","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.20073","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:06528a3730f0f2593c4aacb630e4f0674d4087c634c4af3c878f366c086fcc5d","observation_id":"7711b20f-5074-4cc0-b9e5-82086d47af10","resolution":{"observed_at":"2026-05-13T07:13:34.687091Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.12538","last_updated":"2026-01-18T18:58:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-18T18:58:23Z","title":"Agentic Reasoning for Large Language Models","version":1},"cited_work":{"arxiv_id":"2601.12538","doi":"10.48550/arxiv.2601.12538","metadata_source":"pith","pith_arxiv_id":"2601.12538","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Agentic Reasoning for Large Language Models","venue":"cs.AI","work_id":"062546cd-e1a7-46e4-b617-c6b8e19b6fa3","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.12538","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e205efce3c3921caa19b710ad7cb1918a29f27dc1a05488eff323e34419872a3","observation_id":"f9d25609-d38b-42b2-8332-d678f8004c54","resolution":{"observed_at":"2026-05-17T15:14:26.657956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T10:08:10.217033+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T10:08:10.217033+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T10:08:10.217033+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07572","last_updated":"2025-08-10T05:59:20Z","snapshot_observed_at":"2026-07-06T20:20:33.068936Z","submitted_at":"2025-01-13T18:58:07Z","title":"WebWalker: Benchmarking LLMs in Web Traversal","version":3},"cited_work":{"arxiv_id":"2501.07572","doi":"10.48550/arxiv.2501.07572","metadata_source":"pith","pith_arxiv_id":"2501.07572","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Jialong Wu, Baixuan Li, Runnan Fang, Wenbiao Yin, Liwen Zhang, Zhengwei Tao, Dingchu Zhang, Zekun Xi, Gang Fu, Yong Jiang, Pengjun Xie, Fei Huang, and Jingren Zhou","venue":"cs.CL","work_id":"8528e4cd-bcbc-4f57-9bd2-11ce86dd3493","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2501.07572","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1058a54ef14faca80041716eb290694b4f7e8a1b6239774ee63e70013d1b1865","observation_id":"f4c1fd2f-5f2e-44e6-af5b-fb0dce07b088","resolution":{"observed_at":"2026-05-13T05:07:17.791143Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.01223","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T16:24:01.012815Z","title":"Learn hard problems during rl with reference guided fine-tuning","venue":null,"work_id":"cca99282-a0f6-4e3b-894f-0589afedb0de","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:04ca2c7cce28f563509b4b194643decf6bd84770a78ceeb62d75a6ccc11f56a1","observation_id":"82846ed5-0a12-479c-a3b3-b13b3b5eddd7","resolution":{"observed_at":"2026-05-13T05:07:17.702708Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":"8866ac21-13df-424b-a428-ab1ffff49b2e","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:bdf58b388079b5335e3166e2d069c37280b5b027e380c26a098fc8f2eb77be93","observation_id":"a51275f2-6cbc-4d04-9b51-8f2c2dfe705d","resolution":{"observed_at":"2026-05-13T11:07:39.808900Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14945","last_updated":"2025-06-22T00:18:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-21T08:09:13Z","title":"Learning to Reason under Off-Policy Guidance","version":5},"cited_work":{"arxiv_id":"2504.14945","doi":"10.48550/arxiv.2504.14945","metadata_source":"pith","pith_arxiv_id":"2504.14945","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Learning to Reason under Off-Policy Guidance","venue":"cs.LG","work_id":"4ebcdbe2-5000-4f58-a7e0-aa9ae381b684","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.14945","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:206b610b85c2a00fc59a7c4744502072e65183decfeb42b167d1857dcc93d49c","observation_id":"1d69ddf5-36fb-4f9f-ae7f-c240932e4ed3","resolution":{"observed_at":"2026-05-15T23:17:03.075876Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.22190","last_updated":"2026-05-25T21:32:44Z","snapshot_observed_at":"2026-08-01T20:57:39.795956Z","submitted_at":"2026-02-25T18:34:57Z","title":"GUI-Libra: Training Native GUI Agents to Reason and Act with Action-aware Supervision and Partially Verifiable RL","version":2},"cited_work":{"arxiv_id":"2602.22190","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.22190","snapshot_observed_at":"2026-07-02T11:56:55.516310Z","title":"Gui-libra: Training native gui agents to reason and act with action-aware supervision and partially verifiable rl","venue":"cs.LG","work_id":"92549a20-b67b-4432-a2f1-81176c3e0d06","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2602.22190","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:590db65eb7b6c7ae6e82bc31b0f1be5385c88b76893b20c82565c3010497513a","observation_id":"64b3bdc3-82c7-4ed2-8889-7e699ce8bea5","resolution":{"observed_at":"2026-05-27T02:04:34.683381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T18:37:32.087006Z","title":"React: Synergizing reasoning and acting in language models","venue":null,"work_id":"404308e4-3b7a-4845-8899-d58a0072d6ed","year":2022},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e71e3e7fdc4e905d1da6333b3bbc2d763f33fc07c23c743452cc2e4944aa03f9","observation_id":"9f6a887b-9d47-4c88-ae0f-10733a8b4686","resolution":{"observed_at":"2026-05-13T11:07:39.827410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-07-11T01:37:42.734447Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:ad2bbd3fefa1040ebdda521094ebf51ceabd18baf16cbaf226cd892ad11a4c9f","observation_id":"afad9ae9-cc3b-4580-a25f-ea603716bab3","resolution":{"observed_at":"2026-05-13T05:07:17.666389Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.03048","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T04:27:36.740990Z","title":"Coba-rl: Capability-oriented budget allocation for reinforcement learning in llms","venue":null,"work_id":"dfc92b4c-afc7-4195-b7d6-d28f98ea47f6","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:dddab5f5d7c79c534ab38639286d87c75ae966d1dab4279537df86ec30da9990","observation_id":"69d3835c-60f0-442b-9165-e06fad3beb15","resolution":{"observed_at":"2026-05-13T05:07:17.669790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.21383","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2603.21383 , year=","venue":null,"work_id":"97f6da2e-0071-494f-976b-8422ffc48b69","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:3efaeb34573863548950d03cd1e588e4eab2c8296703f24a52a03c4ff1dc1b5e","observation_id":"7f6c0cb2-ad96-4afd-8d81-644cb60a574a","resolution":{"observed_at":"2026-05-13T05:07:17.663444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.14880","last_updated":"2025-09-01T15:33:47Z","snapshot_observed_at":"2026-07-06T22:15:41.147169Z","submitted_at":"2025-08-20T17:51:20Z","title":"MedResearcher-R1: Expert-Level Medical Deep Researcher via A Knowledge-Informed Trajectory Synthesis Framework","version":3},"cited_work":{"arxiv_id":"2508.14880","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.14880","snapshot_observed_at":"2026-06-29T21:33:58.797000Z","title":"Medresearcher-r1: Expert-level medical deep researcher via a knowledge-informed trajectory synthesis framework","venue":null,"work_id":"adb64f70-9b5d-47c5-b54f-0a5eb958a55b","year":2018},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2508.14880","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:d9c0ebadecb9201e70df250466bd418ca620f890b7200b5b0773befb4c7f4331","observation_id":"41b76d43-58f0-4bec-a003-97980b44661c","resolution":{"observed_at":"2026-05-13T05:07:17.656657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.14476","last_updated":"2025-05-20T01:37:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-18T17:49:06Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","version":2},"cited_work":{"arxiv_id":"2503.14476","doi":"10.48550/arxiv.2503.14476","metadata_source":"pith","pith_arxiv_id":"2503.14476","snapshot_observed_at":"2026-07-11T03:07:50.815080Z","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","venue":"cs.LG","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2503.14476","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:245ffcda86fea8fb826c0a6291ede6205cc2543df432630a044e97d6f710557b","observation_id":"1c777d36-c31c-4b72-bea6-5de65d9332bc","resolution":{"observed_at":"2026-05-13T05:07:17.694521Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T09:23:06.254602+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":"2504.13837","doi":"10.48550/arxiv.2504.13837","metadata_source":"pith","pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-07-10T22:47:36.894457Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","venue":"cs.AI","work_id":"d854765a-e664-41c0-8655-21c4bf2e0cc4","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:37a64394db1ca837c4f57d9a7c2a720484a2ad205716ee48e3efda66fd629b55","observation_id":"66a6d182-aa4a-4d38-b0c8-a96694d6572c","resolution":{"observed_at":"2026-05-13T05:07:17.659881Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.10395","doi":"10.48550/arxiv.2511.10395","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Agentevolver: Towards efficient self-evolving agent system","venue":null,"work_id":"e9710c06-579f-415b-88ba-965ce465b757","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:572a2fdcd746710c9b9778901c2b3ecac46aab27ef44d22a9e627f5b99c2d4a8","observation_id":"a850c3b3-a2dc-4ad3-8e8c-acf481ef8173","resolution":{"observed_at":"2026-05-13T05:07:17.785140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"cited_work":{"arxiv_id":"2509.02547","doi":"10.48550/arxiv.2509.02547","metadata_source":"pith","pith_arxiv_id":"2509.02547","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","venue":"cs.AI","work_id":"87909127-da20-4ccc-8ae3-4a4a20ef81b7","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2509.02547","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:e70fb7c4babf289678394d600a29640b86252de57094394c0ab8adde473c1d4d","observation_id":"ff787be5-ad14-4556-bcbd-6f07b151209e","resolution":{"observed_at":"2026-05-13T05:07:17.608879Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.11408","doi":"10.48550/arxiv.2508.11408","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"arXiv preprint arXiv:2508.11408 , year=","venue":null,"work_id":"994df3c4-5dc6-45c0-9db5-e82e415ce5d8","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:bef38348fec72ed7c26cdb4ffd0f48f9ca9f698b209e4e09ce84563e18c902be","observation_id":"5eddc1e9-79b6-4ecb-998d-34d290444f99","resolution":{"observed_at":"2026-05-13T05:07:17.705737Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.18734","last_updated":"2026-03-20T15:40:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-26T17:56:50Z","title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","version":3},"cited_work":{"arxiv_id":"2601.18734","doi":"10.18653/v1/2025.emnlp-main.125.https://aclanthology.org/2025.emnlp-main.125/","metadata_source":"pith","pith_arxiv_id":"2601.18734","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","venue":"cs.LG","work_id":"bae00e84-9b0d-433d-a066-20b951f0b4d0","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2601.18734","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:ace9aa4da2963be9a6719f945118a63d6633f6d71523a4e9509d655a94e253d3","observation_id":"d45a8540-4c29-4f0d-8030-aef3cc060ce8","resolution":{"observed_at":"2026-05-13T05:07:17.779421Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.01161","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T22:26:37.214162Z","title":"Prosperity before collapse: How far can off-policy rl reach with stale data on llms?","venue":null,"work_id":"99a02056-55bc-45e6-8d10-7f31cf541473","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:c4316b021c3599193bb4aeb421fc3f4216ef66c4ddfb61e79d1e9a28f5ee57a5","observation_id":"4bd1baa4-5550-45eb-b3a4-96e166e18099","resolution":{"observed_at":"2026-05-13T05:07:17.708884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.09856","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:09:59.313952Z","title":"Code2world: A gui world model via renderable code generation","venue":null,"work_id":"9ee28d94-edc8-4660-81eb-40b5d9e02315","year":2026},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:6908407ca6b1fb80427dd2ee19f907fc7a006559f46f3fbdf66503bc91bbc2e8","observation_id":"379ba4ab-3a65-469a-bd7e-d9cdcdfc50cf","resolution":{"observed_at":"2026-05-13T05:07:17.639387Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.07911","last_updated":"2023-11-14T05:13:55Z","snapshot_observed_at":"2026-07-06T16:47:08.877195Z","submitted_at":"2023-11-14T05:13:55Z","title":"Instruction-Following Evaluation for Large Language Models","version":1},"cited_work":{"arxiv_id":"2311.07911","doi":"10.48550/arxiv.2311.07911","metadata_source":"pith","pith_arxiv_id":"2311.07911","snapshot_observed_at":"2026-07-10T12:07:03.672931Z","title":"Instruction-Following Evaluation for Large Language Models","venue":"cs.CL","work_id":"3aa06177-125a-4f5a-8f4a-8070c5986c26","year":2023},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2311.07911","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:4fc2bdc4435f063bdaa705368690b154cd7eab84ba6184baa58c8e3b0d0a4bb2","observation_id":"66043e16-8a57-43b6-9729-a26b0da6908c","resolution":{"observed_at":"2026-05-13T05:07:17.642066Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19314","last_updated":"2025-05-01T05:02:57Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-27T17:32:43Z","title":"BrowseComp-ZH: Benchmarking Web Browsing Ability of Large Language Models in Chinese","version":2},"cited_work":{"arxiv_id":"2504.19314","doi":"10.48550/arxiv.2504.19314","metadata_source":"pith","pith_arxiv_id":"2504.19314","snapshot_observed_at":"2026-07-10T17:27:26.673943Z","title":"BrowseComp-ZH: Benchmarking Web Browsing Ability of Large Language Models in Chinese","venue":"cs.CL","work_id":"d2997896-a54e-43f5-80ac-d4d548116d21","year":2025},"citing_paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-13T05:02:49.206053Z"},"links":{"cited_paper":"/paper/2504.19314","citing_paper":"/paper/2605.12004"},"observation_digest":"sha256:1409fac0552fe6f62e45a7cc2147b4eb3eaf331410ade6c87bb9f3d71c9b15e0","observation_id":"3bb80b65-9f3d-4e17-a552-07be46bb0b8d","resolution":{"observed_at":"2026-05-17T22:04:50.009871Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-01T06:32:01.292127+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.12004","last_updated":"2026-05-12T11:54:23Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T11:54:23Z","title":"Learning Agentic Policy from Action Guidance"},"reference_resolution":{"displayed":82,"state_counts":{"malformed_identifier":0,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":1,"verified_exact":56,"verified_fuzzy":21},"total_outbound_references":82},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-01T06:32:01.292127+00:00","source":"crossref"},{"observed_at":"2026-08-01T06:31:58.492377+00:00","source":"retraction_watch"}],"thesis":"As of 1 August 2026, this Paper Citation Record lists 82 of 82 outbound references and 1 inbound Pith citation observation for arXiv:2605.12004."}