{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:EU7F35APYWV6JRLHP5AFX3RVIB","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"689c55c3c5ae3aca2c60b3f3540ffacc1b2a680b0d02aea526eca225a4cf2841","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-20T02:09:07Z","title_canon_sha256":"71a48b5014df09b99dd68e345bb01df2692834e429f0607587faef811aad2f0e"},"schema_version":"1.0","source":{"id":"2405.11727","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.11727","created_at":"2026-07-05T09:57:42Z"},{"alias_kind":"arxiv_version","alias_value":"2405.11727v2","created_at":"2026-07-05T09:57:42Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.11727","created_at":"2026-07-05T09:57:42Z"},{"alias_kind":"pith_short_12","alias_value":"EU7F35APYWV6","created_at":"2026-07-05T09:57:42Z"},{"alias_kind":"pith_short_16","alias_value":"EU7F35APYWV6JRLH","created_at":"2026-07-05T09:57:42Z"},{"alias_kind":"pith_short_8","alias_value":"EU7F35AP","created_at":"2026-07-05T09:57:42Z"}],"graph_snapshots":[{"event_id":"sha256:139147d2743a1f7e3e57495e265a6f89c57dcaad081035a0622a5ecd5a72e2c8","target":"graph","created_at":"2026-07-05T09:57:42Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.11727/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning (RL) algorithms often struggle with low training efficiency. A common approach to address this challenge is integrating model-based planning algorithms, such as Monte Carlo Tree Search (MCTS) or Value Iteration (VI), into the environmental model. However, VI requires iterating over a large tensor which updates the value of the preceding state based on the succeeding state through value propagation, resulting in computationally intensive operations. To enhance the RL training efficiency, we propose improving the efficiency of the value learning process. In deterministic e","authors_text":"Dong Gong, Javen Q. Shi, Stefano V. Albrecht, Zhen Zhang, Zidu Yin","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-20T02:09:07Z","title":"Highway Graph to Accelerate Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.11727","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4a52ddef11ec3c1c76e512e3b220ab715b574ed941b56e95706a636d4de60d07","target":"record","created_at":"2026-07-05T09:57:42Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"689c55c3c5ae3aca2c60b3f3540ffacc1b2a680b0d02aea526eca225a4cf2841","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-20T02:09:07Z","title_canon_sha256":"71a48b5014df09b99dd68e345bb01df2692834e429f0607587faef811aad2f0e"},"schema_version":"1.0","source":{"id":"2405.11727","kind":"arxiv","version":2}},"canonical_sha256":"253e5df40fc5abe4c5677f405bee35407539b207d6f2cef14baca067d1267e26","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"253e5df40fc5abe4c5677f405bee35407539b207d6f2cef14baca067d1267e26","first_computed_at":"2026-07-05T09:57:42.689239Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:57:42.689239Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"d58oktSUgrffDepNdF4fAFoA9XZBVVeUNRkj/6WrT/v14kak+K+fAI5W+ehyB0pXyMT7pOiMKmpsyrhFSiARAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:57:42.689739Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.11727","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4a52ddef11ec3c1c76e512e3b220ab715b574ed941b56e95706a636d4de60d07","sha256:139147d2743a1f7e3e57495e265a6f89c57dcaad081035a0622a5ecd5a72e2c8"],"state_sha256":"136839853b1bec36de84deecf91d8b2def85cc482ae31086ea1c4cd6f9ec5290"}