{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:EY5EIUT4X3SFKTPH63GO5F4ZQC","short_pith_number":"pith:EY5EIUT4","canonical_record":{"source":{"id":"2505.20737","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-27T05:27:54Z","cross_cats_sorted":[],"title_canon_sha256":"62ee88d7425d9bc20f9da6dfa9ec78100838bb5cec39817a1bfc61d792d46674","abstract_canon_sha256":"268aef29cf02a3839d10b203d31d8ff4136389b0e43ef1e6b1c5e94f421d16b4"},"schema_version":"1.0"},"canonical_sha256":"263a44527cbee4554de7f6ccee979980a1886ec3e0731ab8c18ab18be9e2cb17","source":{"kind":"arxiv","id":"2505.20737","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.20737","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"arxiv_version","alias_value":"2505.20737v1","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20737","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"pith_short_12","alias_value":"EY5EIUT4X3SF","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"pith_short_16","alias_value":"EY5EIUT4X3SFKTPH","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"pith_short_8","alias_value":"EY5EIUT4","created_at":"2026-07-05T11:10:17Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:EY5EIUT4X3SFKTPH63GO5F4ZQC","target":"record","payload":{"canonical_record":{"source":{"id":"2505.20737","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-27T05:27:54Z","cross_cats_sorted":[],"title_canon_sha256":"62ee88d7425d9bc20f9da6dfa9ec78100838bb5cec39817a1bfc61d792d46674","abstract_canon_sha256":"268aef29cf02a3839d10b203d31d8ff4136389b0e43ef1e6b1c5e94f421d16b4"},"schema_version":"1.0"},"canonical_sha256":"263a44527cbee4554de7f6ccee979980a1886ec3e0731ab8c18ab18be9e2cb17","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:17.742963Z","signature_b64":"Vluzv3zp18c0MybND0bmRDn3PJjVUwMdD4ITxC+iqlMgWUnvKIwjOPAhZ1oUmpXYA0fHtgB/Rw3FtKwfdb3+Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"263a44527cbee4554de7f6ccee979980a1886ec3e0731ab8c18ab18be9e2cb17","last_reissued_at":"2026-07-05T11:10:17.742474Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:17.742474Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.20737","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:10:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Iv9BqIpe0sRAR8GE76j0/XzTi3+7ruUlIcxHaJvfH/0Q84dART6f85ABK0E/WfTlw59wwGscJ5AuN9bLI+LBBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T21:05:05.680287Z"},"content_sha256":"aab411244c5200be952e0419c2bb77bfae296a96db9dea626961f024acae0c3c","schema_version":"1.0","event_id":"sha256:aab411244c5200be952e0419c2bb77bfae296a96db9dea626961f024acae0c3c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:EY5EIUT4X3SFKTPH63GO5F4ZQC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RRO: LLM Agent Optimization Through Rising Reward Trajectories","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Haoming Jiang, Jingbo Shang, Jingfeng Yang, Samarth Varshney, Sheikh Muhammad Sarwar, Sreyashi Nag, Xianfeng Tang, Zilong Wang","submitted_at":"2025-05-27T05:27:54Z","abstract_excerpt":"Large language models (LLMs) have exhibited extraordinary performance in a variety of tasks while it remains challenging for them to solve complex multi-step tasks as agents. In practice, agents sensitive to the outcome of certain key steps which makes them likely to fail the task because of a subtle mistake in the planning trajectory. Recent approaches resort to calibrating the reasoning process through reinforcement learning. They reward or penalize every reasoning step with process supervision, as known as Process Reward Models (PRMs). However, PRMs are difficult and costly to scale up with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20737","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20737/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:10:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jZYZMPse47T32yVtUNnquZYUBaRyq9BWuOTo2H4AlHRF4cHeLCRdhwkgqq/AAn0cjCeweMbKNMhloKVDUE2QBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T21:05:05.680770Z"},"content_sha256":"858e9c1b96629d75e76922312082a3fc9341c5e9a631f81b36d1f6bb906e57cf","schema_version":"1.0","event_id":"sha256:858e9c1b96629d75e76922312082a3fc9341c5e9a631f81b36d1f6bb906e57cf"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC/bundle.json","state_url":"https://pith.science/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T21:05:05Z","links":{"resolver":"https://pith.science/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC","bundle":"https://pith.science/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC/bundle.json","state":"https://pith.science/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/EY5EIUT4X3SFKTPH63GO5F4ZQC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:EY5EIUT4X3SFKTPH63GO5F4ZQC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"268aef29cf02a3839d10b203d31d8ff4136389b0e43ef1e6b1c5e94f421d16b4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-27T05:27:54Z","title_canon_sha256":"62ee88d7425d9bc20f9da6dfa9ec78100838bb5cec39817a1bfc61d792d46674"},"schema_version":"1.0","source":{"id":"2505.20737","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.20737","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"arxiv_version","alias_value":"2505.20737v1","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20737","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"pith_short_12","alias_value":"EY5EIUT4X3SF","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"pith_short_16","alias_value":"EY5EIUT4X3SFKTPH","created_at":"2026-07-05T11:10:17Z"},{"alias_kind":"pith_short_8","alias_value":"EY5EIUT4","created_at":"2026-07-05T11:10:17Z"}],"graph_snapshots":[{"event_id":"sha256:858e9c1b96629d75e76922312082a3fc9341c5e9a631f81b36d1f6bb906e57cf","target":"graph","created_at":"2026-07-05T11:10:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.20737/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models (LLMs) have exhibited extraordinary performance in a variety of tasks while it remains challenging for them to solve complex multi-step tasks as agents. In practice, agents sensitive to the outcome of certain key steps which makes them likely to fail the task because of a subtle mistake in the planning trajectory. Recent approaches resort to calibrating the reasoning process through reinforcement learning. They reward or penalize every reasoning step with process supervision, as known as Process Reward Models (PRMs). However, PRMs are difficult and costly to scale up with","authors_text":"Haoming Jiang, Jingbo Shang, Jingfeng Yang, Samarth Varshney, Sheikh Muhammad Sarwar, Sreyashi Nag, Xianfeng Tang, Zilong Wang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-27T05:27:54Z","title":"RRO: LLM Agent Optimization Through Rising Reward Trajectories"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20737","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:aab411244c5200be952e0419c2bb77bfae296a96db9dea626961f024acae0c3c","target":"record","created_at":"2026-07-05T11:10:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"268aef29cf02a3839d10b203d31d8ff4136389b0e43ef1e6b1c5e94f421d16b4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-27T05:27:54Z","title_canon_sha256":"62ee88d7425d9bc20f9da6dfa9ec78100838bb5cec39817a1bfc61d792d46674"},"schema_version":"1.0","source":{"id":"2505.20737","kind":"arxiv","version":1}},"canonical_sha256":"263a44527cbee4554de7f6ccee979980a1886ec3e0731ab8c18ab18be9e2cb17","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"263a44527cbee4554de7f6ccee979980a1886ec3e0731ab8c18ab18be9e2cb17","first_computed_at":"2026-07-05T11:10:17.742474Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:10:17.742474Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Vluzv3zp18c0MybND0bmRDn3PJjVUwMdD4ITxC+iqlMgWUnvKIwjOPAhZ1oUmpXYA0fHtgB/Rw3FtKwfdb3+Aw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:10:17.742963Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.20737","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:aab411244c5200be952e0419c2bb77bfae296a96db9dea626961f024acae0c3c","sha256:858e9c1b96629d75e76922312082a3fc9341c5e9a631f81b36d1f6bb906e57cf"],"state_sha256":"b681e20729ec3ea0a9078b779b4ac9d79bdeecf648b02478036e57c14e31f453"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pP/KFTwYKbxi1/WbeLVQebo2tFIkF8Hln+lXlsWgebb9LMOpwOLVPYmD2MOd3c1JBVOQ2lMggegttmXj23WNDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T21:05:05.685325Z","bundle_sha256":"236ef85e3e4d2c11adcf5a96b2be7e370b7b6170b44e8805d481d7dd107d29fb"}}