{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:XBLQGQCOCAIYBFDCW7IRQTJX6G","short_pith_number":"pith:XBLQGQCO","canonical_record":{"source":{"id":"2607.04963","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-06T11:50:06Z","cross_cats_sorted":[],"title_canon_sha256":"64ac0ec6c4a6cb4c28d89106c0145cec07e5cdcea7881a09a343dca888650c09","abstract_canon_sha256":"4719fdb437e1cefa9cb2e53a665aa28d6edb5cb3102ccb323bb60b93b2364179"},"schema_version":"1.0"},"canonical_sha256":"b85703404e1011809462b7d1184d37f1b031a4a269b485d1b90d5b6fd732f206","source":{"kind":"arxiv","id":"2607.04963","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.04963","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"arxiv_version","alias_value":"2607.04963v1","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.04963","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"pith_short_12","alias_value":"XBLQGQCOCAIY","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"pith_short_16","alias_value":"XBLQGQCOCAIYBFDC","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"pith_short_8","alias_value":"XBLQGQCO","created_at":"2026-07-07T02:20:15Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:XBLQGQCOCAIYBFDCW7IRQTJX6G","target":"record","payload":{"canonical_record":{"source":{"id":"2607.04963","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-06T11:50:06Z","cross_cats_sorted":[],"title_canon_sha256":"64ac0ec6c4a6cb4c28d89106c0145cec07e5cdcea7881a09a343dca888650c09","abstract_canon_sha256":"4719fdb437e1cefa9cb2e53a665aa28d6edb5cb3102ccb323bb60b93b2364179"},"schema_version":"1.0"},"canonical_sha256":"b85703404e1011809462b7d1184d37f1b031a4a269b485d1b90d5b6fd732f206","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:20:15.546613Z","signature_b64":"5Eje1NBIufD8Ed9qaHjcKQLV4L7nYvh8pL0pJeaY48QXj+bnMpFXGNzz1R3dqYmHsH6y/YJK4h8zHge8e69TDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b85703404e1011809462b7d1184d37f1b031a4a269b485d1b90d5b6fd732f206","last_reissued_at":"2026-07-07T02:20:15.545670Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:20:15.545670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.04963","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T02:20:15Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Wi/B/IAtIDblJkRX7XzImSIGzyx/m2vrI+Ka7pminEAPbHIZfhUrK86UfnpkUjJz7LR/ZR/9Q+uAQNtRPYQrBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-27T16:02:35.910552Z"},"content_sha256":"fb9127908bf41d7e4f3ddfd5b5f174380ebc116d328f80e8fb73e5e27886c30d","schema_version":"1.0","event_id":"sha256:fb9127908bf41d7e4f3ddfd5b5f174380ebc116d328f80e8fb73e5e27886c30d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:XBLQGQCOCAIYBFDCW7IRQTJX6G","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dongnan Liu, Feng Zhang, Jie Liu, Jinjian Zhang, Linjian Mo, Ming Kong, Mutian Bao, Qiang Zhu, Qiuyi Qi, Tian Liang, Wei Zhou","submitted_at":"2026-07-06T11:50:06Z","abstract_excerpt":"Reinforcement Learning (RL) is the dominant paradigm for training Large Language Model (LLM) agents on long-horizon tasks. However, sparse and delayed rewards often lead to trajectory neglect, in which agents lose focus on the task goal and interaction history at intermediate steps. Prior work has explored step-level supervision using Shannon-entropy-based uncertainty signals, which conflate inherent state complexity with agent confidence and therefore provide unreliable estimates of decision reliability. To address this issue, we propose normalized entropy, which measures confidence deviation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.04963","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.04963/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T02:20:15Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4zT4QcOFsoj11/T342QQEU2ezDtrHzYxetFQ1dcdX7RqPdRzoIMniGv0MT1LSHxdUaaWkVFANB+GhqCam8IrBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-27T16:02:35.911109Z"},"content_sha256":"f1f1a4c844358e5c308998cda689b0a48071d9dbac7164c15ace45ab1702311c","schema_version":"1.0","event_id":"sha256:f1f1a4c844358e5c308998cda689b0a48071d9dbac7164c15ace45ab1702311c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G/bundle.json","state_url":"https://pith.science/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-27T16:02:35Z","links":{"resolver":"https://pith.science/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G","bundle":"https://pith.science/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G/bundle.json","state":"https://pith.science/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XBLQGQCOCAIYBFDCW7IRQTJX6G/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:XBLQGQCOCAIYBFDCW7IRQTJX6G","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4719fdb437e1cefa9cb2e53a665aa28d6edb5cb3102ccb323bb60b93b2364179","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-06T11:50:06Z","title_canon_sha256":"64ac0ec6c4a6cb4c28d89106c0145cec07e5cdcea7881a09a343dca888650c09"},"schema_version":"1.0","source":{"id":"2607.04963","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.04963","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"arxiv_version","alias_value":"2607.04963v1","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.04963","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"pith_short_12","alias_value":"XBLQGQCOCAIY","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"pith_short_16","alias_value":"XBLQGQCOCAIYBFDC","created_at":"2026-07-07T02:20:15Z"},{"alias_kind":"pith_short_8","alias_value":"XBLQGQCO","created_at":"2026-07-07T02:20:15Z"}],"graph_snapshots":[{"event_id":"sha256:f1f1a4c844358e5c308998cda689b0a48071d9dbac7164c15ace45ab1702311c","target":"graph","created_at":"2026-07-07T02:20:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.04963/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) is the dominant paradigm for training Large Language Model (LLM) agents on long-horizon tasks. However, sparse and delayed rewards often lead to trajectory neglect, in which agents lose focus on the task goal and interaction history at intermediate steps. Prior work has explored step-level supervision using Shannon-entropy-based uncertainty signals, which conflate inherent state complexity with agent confidence and therefore provide unreliable estimates of decision reliability. To address this issue, we propose normalized entropy, which measures confidence deviation","authors_text":"Dongnan Liu, Feng Zhang, Jie Liu, Jinjian Zhang, Linjian Mo, Ming Kong, Mutian Bao, Qiang Zhu, Qiuyi Qi, Tian Liang, Wei Zhou","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-06T11:50:06Z","title":"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.04963","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fb9127908bf41d7e4f3ddfd5b5f174380ebc116d328f80e8fb73e5e27886c30d","target":"record","created_at":"2026-07-07T02:20:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4719fdb437e1cefa9cb2e53a665aa28d6edb5cb3102ccb323bb60b93b2364179","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-06T11:50:06Z","title_canon_sha256":"64ac0ec6c4a6cb4c28d89106c0145cec07e5cdcea7881a09a343dca888650c09"},"schema_version":"1.0","source":{"id":"2607.04963","kind":"arxiv","version":1}},"canonical_sha256":"b85703404e1011809462b7d1184d37f1b031a4a269b485d1b90d5b6fd732f206","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b85703404e1011809462b7d1184d37f1b031a4a269b485d1b90d5b6fd732f206","first_computed_at":"2026-07-07T02:20:15.545670Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-07T02:20:15.545670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5Eje1NBIufD8Ed9qaHjcKQLV4L7nYvh8pL0pJeaY48QXj+bnMpFXGNzz1R3dqYmHsH6y/YJK4h8zHge8e69TDg==","signature_status":"signed_v1","signed_at":"2026-07-07T02:20:15.546613Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.04963","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fb9127908bf41d7e4f3ddfd5b5f174380ebc116d328f80e8fb73e5e27886c30d","sha256:f1f1a4c844358e5c308998cda689b0a48071d9dbac7164c15ace45ab1702311c"],"state_sha256":"6bc0382dfd60b340053cd148fcb1970a0be1a556afd6d86afd608487805dc822"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/Vu8w7T8lR7pssqdibnMa3Bztr8KXuvzZod0CuInYHp5qJU7J8gXecSjxtqFxYdpdG8xsj3FLmLiq15rW6DsDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-27T16:02:35.913421Z","bundle_sha256":"a0ecbf2927f45b1009d4dfcadf69f034a0953cd6d0317271430da3a2b4bf9548"}}