{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WU537KONYGPCLGNMLTPMCGFI5E","short_pith_number":"pith:WU537KON","schema_version":"1.0","canonical_sha256":"b53bbfa9cdc19e2599ac5cdec118a8e927388c6a3d40355eed3b7e3df2081e3b","source":{"kind":"arxiv","id":"2506.20036","version":1},"attestation_state":"computed","paper":{"title":"Hierarchical Reinforcement Learning and Value Optimization for Challenging Quadruped Locomotion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Jeremiah Coholich, Muhammad Ali Murtaza, Seth Hutchinson, Zsolt Kira","submitted_at":"2025-06-24T22:19:15Z","abstract_excerpt":"We propose a novel hierarchical reinforcement learning framework for quadruped locomotion over challenging terrain. Our approach incorporates a two-layer hierarchy in which a high-level policy (HLP) selects optimal goals for a low-level policy (LLP). The LLP is trained using an on-policy actor-critic RL algorithm and is given footstep placements as goals. We propose an HLP that does not require any additional training or environment samples and instead operates via an online optimization process over the learned value function of the LLP. We demonstrate the benefits of this framework by compar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.20036","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-06-24T22:19:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"bd77663eb23e439ded7afcb27333ca886f1f3a8d4418844bb981fe137aca75dd","abstract_canon_sha256":"02fe0d8dd084c928ab08297a299601f1d766204f61625e3308307bfb36f66607"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:59.405044Z","signature_b64":"kMwRk8AlEL+SBOg+yZpFCUVOVL+J6jlpqqcB0yIl0RYzeLCoQURV2NZLWDwJaNQpHxwtaTRNjzY9bNQmtYXdAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b53bbfa9cdc19e2599ac5cdec118a8e927388c6a3d40355eed3b7e3df2081e3b","last_reissued_at":"2026-07-05T11:26:59.404574Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:59.404574Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hierarchical Reinforcement Learning and Value Optimization for Challenging Quadruped Locomotion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Jeremiah Coholich, Muhammad Ali Murtaza, Seth Hutchinson, Zsolt Kira","submitted_at":"2025-06-24T22:19:15Z","abstract_excerpt":"We propose a novel hierarchical reinforcement learning framework for quadruped locomotion over challenging terrain. Our approach incorporates a two-layer hierarchy in which a high-level policy (HLP) selects optimal goals for a low-level policy (LLP). The LLP is trained using an on-policy actor-critic RL algorithm and is given footstep placements as goals. We propose an HLP that does not require any additional training or environment samples and instead operates via an online optimization process over the learned value function of the LLP. We demonstrate the benefits of this framework by compar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.20036","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.20036/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.20036","created_at":"2026-07-05T11:26:59.404631+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.20036v1","created_at":"2026-07-05T11:26:59.404631+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.20036","created_at":"2026-07-05T11:26:59.404631+00:00"},{"alias_kind":"pith_short_12","alias_value":"WU537KONYGPC","created_at":"2026-07-05T11:26:59.404631+00:00"},{"alias_kind":"pith_short_16","alias_value":"WU537KONYGPCLGNM","created_at":"2026-07-05T11:26:59.404631+00:00"},{"alias_kind":"pith_short_8","alias_value":"WU537KON","created_at":"2026-07-05T11:26:59.404631+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E","json":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E.json","graph_json":"https://pith.science/api/pith-number/WU537KONYGPCLGNMLTPMCGFI5E/graph.json","events_json":"https://pith.science/api/pith-number/WU537KONYGPCLGNMLTPMCGFI5E/events.json","paper":"https://pith.science/paper/WU537KON"},"agent_actions":{"view_html":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E","download_json":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E.json","view_paper":"https://pith.science/paper/WU537KON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.20036&json=true","fetch_graph":"https://pith.science/api/pith-number/WU537KONYGPCLGNMLTPMCGFI5E/graph.json","fetch_events":"https://pith.science/api/pith-number/WU537KONYGPCLGNMLTPMCGFI5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E/action/storage_attestation","attest_author":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E/action/author_attestation","sign_citation":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E/action/citation_signature","submit_replication":"https://pith.science/pith/WU537KONYGPCLGNMLTPMCGFI5E/action/replication_record"}},"created_at":"2026-07-05T11:26:59.404631+00:00","updated_at":"2026-07-05T11:26:59.404631+00:00"}