{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Y7HJXN23UUAPLZ4HJL3Y32LJQP","short_pith_number":"pith:Y7HJXN23","schema_version":"1.0","canonical_sha256":"c7ce9bb75ba500f5e7874af78de96983f8cb09aba7bad6eda1ed3de67a36c9ed","source":{"kind":"arxiv","id":"2308.02151","version":3},"attestation_state":"computed","paper":{"title":"Retroformer: Retrospective Large Language Agents with Policy Gradient Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Caiming Xiong, Devansh Arpit, Huan Wang, Jianguo Zhang, Juan Carlos Niebles, Le Xue, Phil Mui, Ran Xu, Rithesh Murthy, Shelby Heinecke, Silvio Savarese, Weiran Yao, Yihao Feng, Zeyuan Chen, Zhiwei Liu","submitted_at":"2023-08-04T06:14:23Z","abstract_excerpt":"Recent months have seen the emergence of a powerful new trend in which large language models (LLMs) are augmented to become autonomous language agents capable of performing objective oriented multi-step tasks on their own, rather than merely responding to queries from human users. Most existing language agents, however, are not optimized using environment-specific rewards. Although some agents enable iterative refinement through verbal feedback, they do not reason and plan in ways that are compatible with gradient-based learning from rewards. This paper introduces a principled framework for re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.02151","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-04T06:14:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f4f107b6e22e521642f3b190cd039b994d0a3ee3421b508b9ed77d3c87ced750","abstract_canon_sha256":"3b83c42db64b140fc516df950e2f969c0c406b75625dcb86e49ad75b48bc9826"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:38.685901Z","signature_b64":"bSgrRtQIlX5UV/Fu8Hj+bLU2H0HxtD5BjiRbSkQLygAs11wpa7DTG0BhgUMJ7hjnR5MUgUHGSup1hPbb1s8vAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7ce9bb75ba500f5e7874af78de96983f8cb09aba7bad6eda1ed3de67a36c9ed","last_reissued_at":"2026-07-05T08:15:38.685298Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:38.685298Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Retroformer: Retrospective Large Language Agents with Policy Gradient Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Caiming Xiong, Devansh Arpit, Huan Wang, Jianguo Zhang, Juan Carlos Niebles, Le Xue, Phil Mui, Ran Xu, Rithesh Murthy, Shelby Heinecke, Silvio Savarese, Weiran Yao, Yihao Feng, Zeyuan Chen, Zhiwei Liu","submitted_at":"2023-08-04T06:14:23Z","abstract_excerpt":"Recent months have seen the emergence of a powerful new trend in which large language models (LLMs) are augmented to become autonomous language agents capable of performing objective oriented multi-step tasks on their own, rather than merely responding to queries from human users. Most existing language agents, however, are not optimized using environment-specific rewards. Although some agents enable iterative refinement through verbal feedback, they do not reason and plan in ways that are compatible with gradient-based learning from rewards. This paper introduces a principled framework for re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.02151","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.02151/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.02151","created_at":"2026-07-05T08:15:38.685373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.02151v3","created_at":"2026-07-05T08:15:38.685373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.02151","created_at":"2026-07-05T08:15:38.685373+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7HJXN23UUAP","created_at":"2026-07-05T08:15:38.685373+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7HJXN23UUAPLZ4H","created_at":"2026-07-05T08:15:38.685373+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7HJXN23","created_at":"2026-07-05T08:15:38.685373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18235","citing_title":"EvolveNav: Proactive Preflection and Self-Evolving Memory for Zero-Shot Object Goal Navigation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07358","citing_title":"A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20477","citing_title":"Training Language Agents to Learn from Experience","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07358","citing_title":"A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15965","citing_title":"From Human Memory to AI Memory: A Survey on Memory Mechanisms in the Era of LLMs","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06130","citing_title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06130","citing_title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07358","citing_title":"A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06130","citing_title":"Skill1: Unified Evolution of Skill-Augmented Agents via Reinforcement Learning","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP","json":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP.json","graph_json":"https://pith.science/api/pith-number/Y7HJXN23UUAPLZ4HJL3Y32LJQP/graph.json","events_json":"https://pith.science/api/pith-number/Y7HJXN23UUAPLZ4HJL3Y32LJQP/events.json","paper":"https://pith.science/paper/Y7HJXN23"},"agent_actions":{"view_html":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP","download_json":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP.json","view_paper":"https://pith.science/paper/Y7HJXN23","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.02151&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7HJXN23UUAPLZ4HJL3Y32LJQP/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7HJXN23UUAPLZ4HJL3Y32LJQP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP/action/storage_attestation","attest_author":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP/action/author_attestation","sign_citation":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP/action/citation_signature","submit_replication":"https://pith.science/pith/Y7HJXN23UUAPLZ4HJL3Y32LJQP/action/replication_record"}},"created_at":"2026-07-05T08:15:38.685373+00:00","updated_at":"2026-07-05T08:15:38.685373+00:00"}