{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:67ECG2CFPLVW3F6ODGWSUEKKBP","short_pith_number":"pith:67ECG2CF","canonical_record":{"source":{"id":"2411.03817","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-06T10:35:11Z","cross_cats_sorted":["cs.CL","cs.HC","cs.RO"],"title_canon_sha256":"66f2c5fa651150644f08ef54c806b38254bc8270980026aeb55903df022f88c7","abstract_canon_sha256":"0f6c3e7772d67443a9b2948b343c6bad131b6bbb6ac6240c208644f659fbb19b"},"schema_version":"1.0"},"canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","source":{"kind":"arxiv","id":"2411.03817","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.03817","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"arxiv_version","alias_value":"2411.03817v3","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03817","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"pith_short_12","alias_value":"67ECG2CFPLVW","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"pith_short_16","alias_value":"67ECG2CFPLVW3F6O","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"pith_short_8","alias_value":"67ECG2CF","created_at":"2026-07-05T09:45:57Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:67ECG2CFPLVW3F6ODGWSUEKKBP","target":"record","payload":{"canonical_record":{"source":{"id":"2411.03817","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-06T10:35:11Z","cross_cats_sorted":["cs.CL","cs.HC","cs.RO"],"title_canon_sha256":"66f2c5fa651150644f08ef54c806b38254bc8270980026aeb55903df022f88c7","abstract_canon_sha256":"0f6c3e7772d67443a9b2948b343c6bad131b6bbb6ac6240c208644f659fbb19b"},"schema_version":"1.0"},"canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:57.316227Z","signature_b64":"xi4VYDk/QzkKhcZBFSu0YgfAJpOdw/F/joHArwMA1mbC8nyLLjCziwLHBtOUTAjP5v8/dbS41QKVTNm+HqULAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","last_reissued_at":"2026-07-05T09:45:57.315639Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:57.315639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2411.03817","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:45:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kOaP/bonkNmejxuAbrplaSNCgLZAvytx1H6LD/+Who/pc1K0BhJWwev7V2Y4eLc6KQgy+Uf1hDVibUWIrVOECQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T00:48:37.263139Z"},"content_sha256":"c93bc5d3f4649fa2e1a050de83357c2a7611ddb884b846309430074a18482086","schema_version":"1.0","event_id":"sha256:c93bc5d3f4649fa2e1a050de83357c2a7611ddb884b846309430074a18482086"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:67ECG2CFPLVW3F6ODGWSUEKKBP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"From Novice to Expert: LLM Agent Policy Optimization via Step-wise Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.HC","cs.RO"],"primary_cat":"cs.AI","authors_text":"Ji-Rong Wen, Mang Wang, Ruibin Xiong, Weipeng Chen, Yutao Zhu, Zhicheng Dou, Zhirui Deng","submitted_at":"2024-11-06T10:35:11Z","abstract_excerpt":"The outstanding capabilities of large language models (LLMs) render them a crucial component in various autonomous agent systems. While traditional methods depend on the inherent knowledge of LLMs without fine-tuning, more recent approaches have shifted toward the reinforcement learning strategy to further enhance agents' ability to solve complex interactive tasks with environments and tools. However, previous approaches are constrained by the sparse reward issue, where existing datasets solely provide a final scalar reward for each multi-step reasoning chain, potentially leading to ineffectiv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03817","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.03817/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:45:57Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Ld3U+Jd+mdNSlsvnvgoQVpKZUTCqvv59f7VOsYUkiyxJ3kykk7JPDDlTkKb7m9G6tVgWNCB3Zeng+T0x+7rDAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T00:48:37.264089Z"},"content_sha256":"95dad764deafdb3795809118f8fe44964224e3f5790a99a6a519697f96723499","schema_version":"1.0","event_id":"sha256:95dad764deafdb3795809118f8fe44964224e3f5790a99a6a519697f96723499"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/bundle.json","state_url":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T00:48:37Z","links":{"resolver":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP","bundle":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/bundle.json","state":"https://pith.science/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/67ECG2CFPLVW3F6ODGWSUEKKBP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:67ECG2CFPLVW3F6ODGWSUEKKBP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0f6c3e7772d67443a9b2948b343c6bad131b6bbb6ac6240c208644f659fbb19b","cross_cats_sorted":["cs.CL","cs.HC","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-06T10:35:11Z","title_canon_sha256":"66f2c5fa651150644f08ef54c806b38254bc8270980026aeb55903df022f88c7"},"schema_version":"1.0","source":{"id":"2411.03817","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2411.03817","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"arxiv_version","alias_value":"2411.03817v3","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.03817","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"pith_short_12","alias_value":"67ECG2CFPLVW","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"pith_short_16","alias_value":"67ECG2CFPLVW3F6O","created_at":"2026-07-05T09:45:57Z"},{"alias_kind":"pith_short_8","alias_value":"67ECG2CF","created_at":"2026-07-05T09:45:57Z"}],"graph_snapshots":[{"event_id":"sha256:95dad764deafdb3795809118f8fe44964224e3f5790a99a6a519697f96723499","target":"graph","created_at":"2026-07-05T09:45:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2411.03817/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The outstanding capabilities of large language models (LLMs) render them a crucial component in various autonomous agent systems. While traditional methods depend on the inherent knowledge of LLMs without fine-tuning, more recent approaches have shifted toward the reinforcement learning strategy to further enhance agents' ability to solve complex interactive tasks with environments and tools. However, previous approaches are constrained by the sparse reward issue, where existing datasets solely provide a final scalar reward for each multi-step reasoning chain, potentially leading to ineffectiv","authors_text":"Ji-Rong Wen, Mang Wang, Ruibin Xiong, Weipeng Chen, Yutao Zhu, Zhicheng Dou, Zhirui Deng","cross_cats":["cs.CL","cs.HC","cs.RO"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-06T10:35:11Z","title":"From Novice to Expert: LLM Agent Policy Optimization via Step-wise Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.03817","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c93bc5d3f4649fa2e1a050de83357c2a7611ddb884b846309430074a18482086","target":"record","created_at":"2026-07-05T09:45:57Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0f6c3e7772d67443a9b2948b343c6bad131b6bbb6ac6240c208644f659fbb19b","cross_cats_sorted":["cs.CL","cs.HC","cs.RO"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-06T10:35:11Z","title_canon_sha256":"66f2c5fa651150644f08ef54c806b38254bc8270980026aeb55903df022f88c7"},"schema_version":"1.0","source":{"id":"2411.03817","kind":"arxiv","version":3}},"canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f7c82368457aeb6d97ce19ad2a114a0bd62f9a71b3328bade84b8c3968da1109","first_computed_at":"2026-07-05T09:45:57.315639Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:45:57.315639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"xi4VYDk/QzkKhcZBFSu0YgfAJpOdw/F/joHArwMA1mbC8nyLLjCziwLHBtOUTAjP5v8/dbS41QKVTNm+HqULAw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:45:57.316227Z","signed_message":"canonical_sha256_bytes"},"source_id":"2411.03817","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c93bc5d3f4649fa2e1a050de83357c2a7611ddb884b846309430074a18482086","sha256:95dad764deafdb3795809118f8fe44964224e3f5790a99a6a519697f96723499"],"state_sha256":"e04ce3c8c82510e0d0782e9c3771a1920a30eb095d40a53d3732dc0f4dc8eac1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"srGdCpB21PbsHswZDXqIxSHvh0/UPOwYYNAcCcX1EMI6PeUsOz5gujbET+YQ39XkHkus6siapbJlJKhmW2hZBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T00:48:37.269963Z","bundle_sha256":"0fe48567b29e7b939fbd36a9cfd94a1254113574aba672f6d50b564b3a964888"}}