{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:TIXCOYNHE7YB7V6RCGZ4OGPWJA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8365a2c704d6f63c4ee5ba4203337c2540a69b986f85b326e1febeedfb1e5a64","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-11T08:50:01Z","title_canon_sha256":"3e797667fc2ee02d7b2d2f1329f8f7c5c553f2c41b3fe190600bf466c81659b5"},"schema_version":"1.0","source":{"id":"2509.09265","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.09265","created_at":"2026-07-05T12:09:32Z"},{"alias_kind":"arxiv_version","alias_value":"2509.09265v1","created_at":"2026-07-05T12:09:32Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09265","created_at":"2026-07-05T12:09:32Z"},{"alias_kind":"pith_short_12","alias_value":"TIXCOYNHE7YB","created_at":"2026-07-05T12:09:32Z"},{"alias_kind":"pith_short_16","alias_value":"TIXCOYNHE7YB7V6R","created_at":"2026-07-05T12:09:32Z"},{"alias_kind":"pith_short_8","alias_value":"TIXCOYNH","created_at":"2026-07-05T12:09:32Z"}],"graph_snapshots":[{"event_id":"sha256:9d91acaf6ea59b41af23ebe640b2f0e293b19d09f5ba87e4d2172529a17ee31e","target":"graph","created_at":"2026-07-05T12:09:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2509.09265/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In long-horizon tasks, recent agents based on Large Language Models (LLMs) face a significant challenge that sparse, outcome-based rewards make it difficult to assign credit to intermediate steps. Previous methods mainly focus on creating dense reward signals to guide learning, either through traditional reinforcement learning techniques like inverse reinforcement learning or by using Process Reward Models for step-by-step feedback. In this paper, we identify a fundamental problem in the learning dynamics of LLMs: the magnitude of policy gradients is inherently coupled with the entropy, which ","authors_text":"Jiacai Liu, Jiawei Wang, Ke Wang, Lin Zhang, Xintao Wang, Yang Wang, Yingru Li, Yuan Lin, Yuqian Fu, Yu Yue","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-11T08:50:01Z","title":"Harnessing Uncertainty: Entropy-Modulated Policy Gradients for Long-Horizon LLM Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09265","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:bfe76c2e3532130eed9a95c04d1aa446e3639de32631290e545b9fa4b30ea767","target":"record","created_at":"2026-07-05T12:09:32Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8365a2c704d6f63c4ee5ba4203337c2540a69b986f85b326e1febeedfb1e5a64","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-11T08:50:01Z","title_canon_sha256":"3e797667fc2ee02d7b2d2f1329f8f7c5c553f2c41b3fe190600bf466c81659b5"},"schema_version":"1.0","source":{"id":"2509.09265","kind":"arxiv","version":1}},"canonical_sha256":"9a2e2761a727f01fd7d111b3c719f6483e60e2aecb02764a7bd84c0aa2ecd824","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9a2e2761a727f01fd7d111b3c719f6483e60e2aecb02764a7bd84c0aa2ecd824","first_computed_at":"2026-07-05T12:09:32.755249Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:09:32.755249Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"mti6w7qeG9jK24GNVri95yFjmtyH4KFrMOd1eGgTPtaA0xqjnsONrQKUB7ZRq0huFQBdNS9kxyvz3lMTbLROBw==","signature_status":"signed_v1","signed_at":"2026-07-05T12:09:32.755815Z","signed_message":"canonical_sha256_bytes"},"source_id":"2509.09265","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:bfe76c2e3532130eed9a95c04d1aa446e3639de32631290e545b9fa4b30ea767","sha256:9d91acaf6ea59b41af23ebe640b2f0e293b19d09f5ba87e4d2172529a17ee31e"],"state_sha256":"d7cb53ae12800e954260ffbd5b962d65d2163cf38417a1f1c56accc71f1c08e7"}