{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:QEPYIMW3JX5KOHLF3JMYLHIGYV","short_pith_number":"pith:QEPYIMW3","canonical_record":{"source":{"id":"2607.27973","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:17:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"df2fe71c444ad950fce862836a1f1ce9fd3b9bb3429d5eeb8871f45b86fbb071","abstract_canon_sha256":"3e6531ff513f3db771d62e51b1ff4e0d62ff1eb96ce034025151905f8b132697"},"schema_version":"1.0"},"canonical_sha256":"811f8432db4dfaa71d65da59859d06c561da263815892e2260b56dc4c07f8295","source":{"kind":"arxiv","id":"2607.27973","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.27973","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"arxiv_version","alias_value":"2607.27973v1","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27973","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"pith_short_12","alias_value":"QEPYIMW3JX5K","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"pith_short_16","alias_value":"QEPYIMW3JX5KOHLF","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"pith_short_8","alias_value":"QEPYIMW3","created_at":"2026-07-31T01:35:09Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:QEPYIMW3JX5KOHLF3JMYLHIGYV","target":"record","payload":{"canonical_record":{"source":{"id":"2607.27973","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:17:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"df2fe71c444ad950fce862836a1f1ce9fd3b9bb3429d5eeb8871f45b86fbb071","abstract_canon_sha256":"3e6531ff513f3db771d62e51b1ff4e0d62ff1eb96ce034025151905f8b132697"},"schema_version":"1.0"},"canonical_sha256":"811f8432db4dfaa71d65da59859d06c561da263815892e2260b56dc4c07f8295","receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"811f8432db4dfaa71d65da59859d06c561da263815892e2260b56dc4c07f8295","last_reissued_at":"2026-07-31T01:35:09.560163Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-31T01:35:09.560163Z"},"source_kind":"arxiv","source_id":"2607.27973","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T01:35:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9tTb9uVzoFKRC2iLrq2ySAcZiV2LWD9BX3p95a18pqmzdIOtRpTLbCCSiknbfEf/tFzn8nJotBppNqnh55TqCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T09:42:44.231655Z"},"content_sha256":"2193b5ec87876898847df4060877c42b769035eed3271393edf2a56fe8f82813","schema_version":"1.0","event_id":"sha256:2193b5ec87876898847df4060877c42b769035eed3271393edf2a56fe8f82813"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:QEPYIMW3JX5KOHLF3JMYLHIGYV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"TAPO: Transition-Aware Policy Optimization for LLM Agents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Cong Li, Peixi Peng, Shudong Liu, Xinyu Hu, Yisen Zhao, Zhan Su, Zhuojian Li","submitted_at":"2026-07-30T10:17:55Z","abstract_excerpt":"Recently, Reinforcement Learning (RL) has emerged as a crucial paradigm for the post-training of Large Language Model (LLM) agents. However, existing methods predominantly rely on sparse task rewards for policy optimization, failing to fully exploit another class of inherently dense supervisory signals naturally present during online interaction: environmental feedback following action execution. Recent theoretical studies suggest that generalization in multi-step, goal-oriented tasks hinges on predictive knowledge of environmental consequences. Inspired by this, we propose TAPO: Transition-Aw"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27973","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27973/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-31T01:35:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dUabfnnqmUmYak2h9xobutsHCIxU92cHPktbshHjN847sN6L85o0QV2yoB7o1R7P9iXtmc0lpDS/lAyrRolwDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T09:42:44.232235Z"},"content_sha256":"f3142b21a240fca1cf53ee3f38b36bc65343e01cd59f7f3f13dc5d40807f1be0","schema_version":"1.0","event_id":"sha256:f3142b21a240fca1cf53ee3f38b36bc65343e01cd59f7f3f13dc5d40807f1be0"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV/bundle.json","state_url":"https://pith.science/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T09:42:44Z","links":{"resolver":"https://pith.science/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV","bundle":"https://pith.science/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV/bundle.json","state":"https://pith.science/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/QEPYIMW3JX5KOHLF3JMYLHIGYV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:QEPYIMW3JX5KOHLF3JMYLHIGYV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3e6531ff513f3db771d62e51b1ff4e0d62ff1eb96ce034025151905f8b132697","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:17:55Z","title_canon_sha256":"df2fe71c444ad950fce862836a1f1ce9fd3b9bb3429d5eeb8871f45b86fbb071"},"schema_version":"1.0","source":{"id":"2607.27973","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.27973","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"arxiv_version","alias_value":"2607.27973v1","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27973","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"pith_short_12","alias_value":"QEPYIMW3JX5K","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"pith_short_16","alias_value":"QEPYIMW3JX5KOHLF","created_at":"2026-07-31T01:35:09Z"},{"alias_kind":"pith_short_8","alias_value":"QEPYIMW3","created_at":"2026-07-31T01:35:09Z"}],"graph_snapshots":[{"event_id":"sha256:f3142b21a240fca1cf53ee3f38b36bc65343e01cd59f7f3f13dc5d40807f1be0","target":"graph","created_at":"2026-07-31T01:35:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.27973/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recently, Reinforcement Learning (RL) has emerged as a crucial paradigm for the post-training of Large Language Model (LLM) agents. However, existing methods predominantly rely on sparse task rewards for policy optimization, failing to fully exploit another class of inherently dense supervisory signals naturally present during online interaction: environmental feedback following action execution. Recent theoretical studies suggest that generalization in multi-step, goal-oriented tasks hinges on predictive knowledge of environmental consequences. Inspired by this, we propose TAPO: Transition-Aw","authors_text":"Cong Li, Peixi Peng, Shudong Liu, Xinyu Hu, Yisen Zhao, Zhan Su, Zhuojian Li","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:17:55Z","title":"TAPO: Transition-Aware Policy Optimization for LLM Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27973","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2193b5ec87876898847df4060877c42b769035eed3271393edf2a56fe8f82813","target":"record","created_at":"2026-07-31T01:35:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3e6531ff513f3db771d62e51b1ff4e0d62ff1eb96ce034025151905f8b132697","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-30T10:17:55Z","title_canon_sha256":"df2fe71c444ad950fce862836a1f1ce9fd3b9bb3429d5eeb8871f45b86fbb071"},"schema_version":"1.0","source":{"id":"2607.27973","kind":"arxiv","version":1}},"canonical_sha256":"811f8432db4dfaa71d65da59859d06c561da263815892e2260b56dc4c07f8295","receipt":{"builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"811f8432db4dfaa71d65da59859d06c561da263815892e2260b56dc4c07f8295","first_computed_at":"2026-07-31T01:35:09.560163Z","kind":"pith_receipt","last_reissued_at":"2026-07-31T01:35:09.560163Z","receipt_version":"0.3","signature_status":"unsigned_v0"},"source_id":"2607.27973","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2193b5ec87876898847df4060877c42b769035eed3271393edf2a56fe8f82813","sha256:f3142b21a240fca1cf53ee3f38b36bc65343e01cd59f7f3f13dc5d40807f1be0"],"state_sha256":"f6dc6fd1ea6a50f84e7909d592c11d14380eea7192d38af1cb6489653a9d87e7"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Q2lYpd64oBux1H4WccHSu9kgAaFyUbRq1DLOXfzPda8hhE7LkhNMsWCKQ+O7xsLGqzJwUtK9U+KOidoIc7rcCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T09:42:44.235875Z","bundle_sha256":"6de29fa20dd882297c97663fb158fd5683457fa0a7e6119919d828ecbd70b928"}}