{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:67ECG2CFPLVW3F6ODGWSUEKKBP","short_pith_number":"pith:67ECG2CF","schema_version":"1.0","canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","source":{"kind":"arxiv","id":"2411.03817","version":3},"attestation_state":"computed","paper":{"title":"From Novice to Expert: LLM Agent Policy Optimization via Step-wise Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.RO"],"primary_cat":"cs.AI","authors_text":"Ji-Rong Wen, Mang Wang, Ruibin Xiong, Weipeng Chen, Yutao Zhu, Zhicheng Dou, Zhirui Deng","submitted_at":"2024-11-06T10:35:11Z","abstract_excerpt":"The outstanding capabilities of large language models (LLMs) render them a crucial component in various autonomous agent systems. While traditional methods depend on the inherent knowledge of LLMs without fine-tuning, more recent approaches have shifted toward the reinforcement learning strategy to further enhance agents' ability to solve complex interactive tasks with environments and tools. However, previous approaches are constrained by the sparse reward issue, where existing datasets solely provide a final scalar reward for each multi-step reasoning chain, potentially leading to ineffectiv"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.03817","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-06T10:35:11Z","cross_cats_sorted":["cs.CL","cs.HC","cs.RO"],"title_canon_sha256":"66f2c5fa651150644f08ef54c806b38254bc8270980026aeb55903df022f88c7","abstract_canon_sha256":"0f6c3e7772d67443a9b2948b343c6bad131b6bbb6ac6240c208644f659fbb19b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:57.316227Z","signature_b64":"xi4VYDk/QzkKhcZBFSu0YgfAJpOdw/F/joHArwMA1mbC8nyLLjCziwLHBtOUTAjP5v8/dbS41QKVTNm+HqULAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","last_reissued_at":"2026-07-05T09:45:57.315639Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:57.315639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From Novice to Expert: LLM Agent Policy Optimization via Step-wise Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.RO"],"primary_cat":"cs.AI","authors_text":"Ji-Rong Wen, Mang Wang, Ruibin Xiong, Weipeng Chen, Yutao Zhu, Zhicheng Dou, Zhirui Deng","submitted_at":"2024-11-06T10:35:11Z","abstract_excerpt":"The outstanding capabilities of large language models (LLMs) render them a crucial component in various autonomous agent systems. While traditional methods depend on the inherent knowledge of LLMs without fine-tuning, more recent approaches have shifted toward the reinforcement learning strategy to further enhance agents' ability to solve complex interactive tasks with environments and tools. However, previous approaches are constrained by the sparse reward issue, where existing datasets solely provide a final scalar reward for each multi-step reasoning chain, potentially leading to ineffectiv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03817","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.03817/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.03817","created_at":"2026-07-05T09:45:57.315703+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.03817v3","created_at":"2026-07-05T09:45:57.315703+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03817","created_at":"2026-07-05T09:45:57.315703+00:00"},{"alias_kind":"pith_short_12","alias_value":"67ECG2CFPLVW","created_at":"2026-07-05T09:45:57.315703+00:00"},{"alias_kind":"pith_short_16","alias_value":"67ECG2CFPLVW3F6O","created_at":"2026-07-05T09:45:57.315703+00:00"},{"alias_kind":"pith_short_8","alias_value":"67ECG2CF","created_at":"2026-07-05T09:45:57.315703+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.09934","citing_title":"TRACER: Verifiable Generative Provenance for Multimodal Tool-Using Agents","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09459","citing_title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16995","citing_title":"SPS: Steering Probability Squeezing for Better Exploration in Reinforcement Learning for Large Language Models","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP","json":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP.json","graph_json":"https://pith.science/api/pith-number/67ECG2CFPLVW3F6ODGWSUEKKBP/graph.json","events_json":"https://pith.science/api/pith-number/67ECG2CFPLVW3F6ODGWSUEKKBP/events.json","paper":"https://pith.science/paper/67ECG2CF"},"agent_actions":{"view_html":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP","download_json":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP.json","view_paper":"https://pith.science/paper/67ECG2CF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.03817&json=true","fetch_graph":"https://pith.science/api/pith-number/67ECG2CFPLVW3F6ODGWSUEKKBP/graph.json","fetch_events":"https://pith.science/api/pith-number/67ECG2CFPLVW3F6ODGWSUEKKBP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/action/storage_attestation","attest_author":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/action/author_attestation","sign_citation":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/action/citation_signature","submit_replication":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/action/replication_record"}},"created_at":"2026-07-05T09:45:57.315703+00:00","updated_at":"2026-07-05T09:45:57.315703+00:00"}