{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XZEC4VZA2B74WLPEDUXQSGER7U","short_pith_number":"pith:XZEC4VZA","schema_version":"1.0","canonical_sha256":"be482e5720d07fcb2de41d2f091891fd2738851688e0a24d4e36ffb0c8545c06","source":{"kind":"arxiv","id":"2501.11425","version":3},"attestation_state":"computed","paper":{"title":"Agent-R: Training Language Model Agents to Reflect via Iterative Self-Training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jiecao Chen, Junjie Ye, Siyu Yuan, Zehui Chen, Zhengyin Du, Zhiheng Xi","submitted_at":"2025-01-20T11:46:04Z","abstract_excerpt":"Large Language Models (LLMs) agents are increasingly pivotal for addressing complex tasks in interactive environments. Existing work mainly focuses on enhancing performance through behavior cloning from stronger experts, yet such approaches often falter in real-world applications, mainly due to the inability to recover from errors. However, step-level critique data is difficult and expensive to collect. Automating and dynamically constructing self-critique datasets is thus crucial to empowering models with intelligent agent capabilities. In this work, we propose an iterative self-training fram"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11425","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2025-01-20T11:46:04Z","cross_cats_sorted":[],"title_canon_sha256":"c9af14af427a15ed0696a16d26a1ec449eb91a4f3844465de1a7551a7ab1c5eb","abstract_canon_sha256":"fb26b0dff099292a23acc24014dcbe48910a591b4909dad189a51f83dac619f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:43.052986Z","signature_b64":"UH7OhXyfsqa2JKdBXQXekfH9DRM7nd8ohdSfKJwrdOB1ZFAVQCilwGej59yQ7KaVbMLAW8duNDRP5Azxc02DAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be482e5720d07fcb2de41d2f091891fd2738851688e0a24d4e36ffb0c8545c06","last_reissued_at":"2026-07-05T10:37:43.052070Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:43.052070Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Agent-R: Training Language Model Agents to Reflect via Iterative Self-Training","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Jiecao Chen, Junjie Ye, Siyu Yuan, Zehui Chen, Zhengyin Du, Zhiheng Xi","submitted_at":"2025-01-20T11:46:04Z","abstract_excerpt":"Large Language Models (LLMs) agents are increasingly pivotal for addressing complex tasks in interactive environments. Existing work mainly focuses on enhancing performance through behavior cloning from stronger experts, yet such approaches often falter in real-world applications, mainly due to the inability to recover from errors. However, step-level critique data is difficult and expensive to collect. Automating and dynamically constructing self-critique datasets is thus crucial to empowering models with intelligent agent capabilities. In this work, we propose an iterative self-training fram"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11425","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11425/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11425","created_at":"2026-07-05T10:37:43.052193+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11425v3","created_at":"2026-07-05T10:37:43.052193+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11425","created_at":"2026-07-05T10:37:43.052193+00:00"},{"alias_kind":"pith_short_12","alias_value":"XZEC4VZA2B74","created_at":"2026-07-05T10:37:43.052193+00:00"},{"alias_kind":"pith_short_16","alias_value":"XZEC4VZA2B74WLPE","created_at":"2026-07-05T10:37:43.052193+00:00"},{"alias_kind":"pith_short_8","alias_value":"XZEC4VZA","created_at":"2026-07-05T10:37:43.052193+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07361","citing_title":"BUS: Brain-Inspired Unsupervised Self-Reflection via Backward Prediction for Multimodal Reasoning","ref_index":85,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00035","citing_title":"Making Failure Safe: A Constrained, Verifiable Agent Framework for Open-Web Data Collection","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02372","citing_title":"COMAP: Co-Evolving World Models and Agent Policies for LLM Agents","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24426","citing_title":"SEAL: Synergistic Co-Evolution of Agents and Learning Environments","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2506.00886","citing_title":"Position: Agent Should Invoke External Tools ONLY When Epistemically Necessary","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2508.08791","citing_title":"Feedback-Driven Tool-Use Improvements in Large Language Models via Automated Build Environments","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15841","citing_title":"MEM1: Learning to Synergize Memory and Reasoning for Efficient Long-Horizon Agents","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10674","citing_title":"Skill-SD: Skill-Conditioned Self-Distillation for Multi-turn LLM Agents","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07774","citing_title":"RoboAgent: Chaining Basic Capabilities for Embodied Task Planning","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06734","citing_title":"TEC: A Collection of Human Trial-and-error Trajectories for Problem Solving","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U","json":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U.json","graph_json":"https://pith.science/api/pith-number/XZEC4VZA2B74WLPEDUXQSGER7U/graph.json","events_json":"https://pith.science/api/pith-number/XZEC4VZA2B74WLPEDUXQSGER7U/events.json","paper":"https://pith.science/paper/XZEC4VZA"},"agent_actions":{"view_html":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U","download_json":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U.json","view_paper":"https://pith.science/paper/XZEC4VZA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11425&json=true","fetch_graph":"https://pith.science/api/pith-number/XZEC4VZA2B74WLPEDUXQSGER7U/graph.json","fetch_events":"https://pith.science/api/pith-number/XZEC4VZA2B74WLPEDUXQSGER7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U/action/storage_attestation","attest_author":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U/action/author_attestation","sign_citation":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U/action/citation_signature","submit_replication":"https://pith.science/pith/XZEC4VZA2B74WLPEDUXQSGER7U/action/replication_record"}},"created_at":"2026-07-05T10:37:43.052193+00:00","updated_at":"2026-07-05T10:37:43.052193+00:00"}