{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RWLFQNTTSNO2WQMA4C72XWDSBL","short_pith_number":"pith:RWLFQNTT","schema_version":"1.0","canonical_sha256":"8d96583673935dab4180e0bfabd8720acb12830a1539d9e12c9efd4b2f2bb567","source":{"kind":"arxiv","id":"2503.08299","version":1},"attestation_state":"computed","paper":{"title":"Distillation-PPO: A Novel Two-Stage Reinforcement Learning Framework for Humanoid Robot Perceptive Locomotion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chenghao Sun, Gang Han, Jiahang Cao, Jiaxu Wang, Jingkai Sun, Qiang Zhang, Renjing Xu, Wen Zhao, Yijie Guo","submitted_at":"2025-03-11T11:10:33Z","abstract_excerpt":"In recent years, humanoid robots have garnered significant attention from both academia and industry due to their high adaptability to environments and human-like characteristics. With the rapid advancement of reinforcement learning, substantial progress has been made in the walking control of humanoid robots. However, existing methods still face challenges when dealing with complex environments and irregular terrains. In the field of perceptive locomotion, existing approaches are generally divided into two-stage methods and end-to-end methods. Two-stage methods first train a teacher policy in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.08299","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-03-11T11:10:33Z","cross_cats_sorted":[],"title_canon_sha256":"ac168402565d8848c57fcddf5a956c0b32788da3ce3b28ea99b942beda0c7437","abstract_canon_sha256":"cff175a06f2f1582f671b1a5e4b24891542be8d64e917830f852785945cf1e52"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:28:56.091585Z","signature_b64":"MD4bQmACabkEoBR8tenXpVbkPYX2upRwMIRVHnYit15Jwqw5H2450jd+wyUQN+tBkbsAEEMeNO0Xpvjfw80BAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d96583673935dab4180e0bfabd8720acb12830a1539d9e12c9efd4b2f2bb567","last_reissued_at":"2026-07-05T10:28:56.091086Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:28:56.091086Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distillation-PPO: A Novel Two-Stage Reinforcement Learning Framework for Humanoid Robot Perceptive Locomotion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Chenghao Sun, Gang Han, Jiahang Cao, Jiaxu Wang, Jingkai Sun, Qiang Zhang, Renjing Xu, Wen Zhao, Yijie Guo","submitted_at":"2025-03-11T11:10:33Z","abstract_excerpt":"In recent years, humanoid robots have garnered significant attention from both academia and industry due to their high adaptability to environments and human-like characteristics. With the rapid advancement of reinforcement learning, substantial progress has been made in the walking control of humanoid robots. However, existing methods still face challenges when dealing with complex environments and irregular terrains. In the field of perceptive locomotion, existing approaches are generally divided into two-stage methods and end-to-end methods. Two-stage methods first train a teacher policy in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.08299","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.08299/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.08299","created_at":"2026-07-05T10:28:56.091150+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.08299v1","created_at":"2026-07-05T10:28:56.091150+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.08299","created_at":"2026-07-05T10:28:56.091150+00:00"},{"alias_kind":"pith_short_12","alias_value":"RWLFQNTTSNO2","created_at":"2026-07-05T10:28:56.091150+00:00"},{"alias_kind":"pith_short_16","alias_value":"RWLFQNTTSNO2WQMA","created_at":"2026-07-05T10:28:56.091150+00:00"},{"alias_kind":"pith_short_8","alias_value":"RWLFQNTT","created_at":"2026-07-05T10:28:56.091150+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.18780","citing_title":"DreamPolicy: A Unified World-model Policy for Scalable Humanoid Locomotion","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06571","citing_title":"Learning Agile Striker Skills for Humanoid Soccer Robots from Noisy Sensory Input","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL","json":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL.json","graph_json":"https://pith.science/api/pith-number/RWLFQNTTSNO2WQMA4C72XWDSBL/graph.json","events_json":"https://pith.science/api/pith-number/RWLFQNTTSNO2WQMA4C72XWDSBL/events.json","paper":"https://pith.science/paper/RWLFQNTT"},"agent_actions":{"view_html":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL","download_json":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL.json","view_paper":"https://pith.science/paper/RWLFQNTT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.08299&json=true","fetch_graph":"https://pith.science/api/pith-number/RWLFQNTTSNO2WQMA4C72XWDSBL/graph.json","fetch_events":"https://pith.science/api/pith-number/RWLFQNTTSNO2WQMA4C72XWDSBL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL/action/storage_attestation","attest_author":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL/action/author_attestation","sign_citation":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL/action/citation_signature","submit_replication":"https://pith.science/pith/RWLFQNTTSNO2WQMA4C72XWDSBL/action/replication_record"}},"created_at":"2026-07-05T10:28:56.091150+00:00","updated_at":"2026-07-05T10:28:56.091150+00:00"}