{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VHLLZOWDUYPNU46DDPUWMBK6LN","short_pith_number":"pith:VHLLZOWD","schema_version":"1.0","canonical_sha256":"a9d6bcbac3a61eda73c31be966055e5b4439e01dcc2ba3c05d148ec572ebbe7b","source":{"kind":"arxiv","id":"2303.03381","version":2},"attestation_state":"computed","paper":{"title":"Real-World Humanoid Locomotion with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Bike Zhang, Ilija Radosavovic, Jitendra Malik, Koushil Sreenath, Tete Xiao, Trevor Darrell","submitted_at":"2023-03-06T18:59:09Z","abstract_excerpt":"Humanoid robots that can autonomously operate in diverse environments have the potential to help address labour shortages in factories, assist elderly at homes, and colonize new planets. While classical controllers for humanoid robots have shown impressive results in a number of settings, they are challenging to generalize and adapt to new environments. Here, we present a fully learning-based approach for real-world humanoid locomotion. Our controller is a causal transformer that takes the history of proprioceptive observations and actions as input and predicts the next action. We hypothesize "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.03381","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2023-03-06T18:59:09Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"22cadf0a7fd2ca5f722bec2af918ec82d25bb8400575476e7672c5a901948b56","abstract_canon_sha256":"4aa82a5f09b44a283aafa63ed560b14202b71b205a00b344c22a54484a189ac3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:23:56.079066Z","signature_b64":"1h/3yPTvDxWWCcWtv8wNSTB8DgSmatm8ntEJKpYHqFONd2IYvG4tvmV3AL8xI7zmIBevwcjZIlZAmcx4yGK/DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a9d6bcbac3a61eda73c31be966055e5b4439e01dcc2ba3c05d148ec572ebbe7b","last_reissued_at":"2026-07-05T07:23:56.078579Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:23:56.078579Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Real-World Humanoid Locomotion with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Bike Zhang, Ilija Radosavovic, Jitendra Malik, Koushil Sreenath, Tete Xiao, Trevor Darrell","submitted_at":"2023-03-06T18:59:09Z","abstract_excerpt":"Humanoid robots that can autonomously operate in diverse environments have the potential to help address labour shortages in factories, assist elderly at homes, and colonize new planets. While classical controllers for humanoid robots have shown impressive results in a number of settings, they are challenging to generalize and adapt to new environments. Here, we present a fully learning-based approach for real-world humanoid locomotion. Our controller is a causal transformer that takes the history of proprioceptive observations and actions as input and predicts the next action. We hypothesize "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.03381","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.03381/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.03381","created_at":"2026-07-05T07:23:56.078637+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.03381v2","created_at":"2026-07-05T07:23:56.078637+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.03381","created_at":"2026-07-05T07:23:56.078637+00:00"},{"alias_kind":"pith_short_12","alias_value":"VHLLZOWDUYPN","created_at":"2026-07-05T07:23:56.078637+00:00"},{"alias_kind":"pith_short_16","alias_value":"VHLLZOWDUYPNU46D","created_at":"2026-07-05T07:23:56.078637+00:00"},{"alias_kind":"pith_short_8","alias_value":"VHLLZOWD","created_at":"2026-07-05T07:23:56.078637+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20645","citing_title":"TACT-ful: Multi-Channel Terrain Affordance and Compliance Training for Payload-Robust Perceptive Humanoid Locomotion","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24592","citing_title":"MuGen: Multi-Skill Generative Locomotion Controller for Humanoid Robots","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07295","citing_title":"Learning Multi-Modal Whole-Body Control for Real-World Humanoid Robots","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01283","citing_title":"The Informational Cost of Agency: A Bounded Measure of Interaction Efficiency for Deployed Reinforcement Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2508.14098","citing_title":"No More Marching: Learning Humanoid Locomotion for Short-Range SE(2) Targets","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01283","citing_title":"The Informational Cost of Agency: A Bounded Measure of Interaction Efficiency for Deployed Reinforcement Learning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN","json":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN.json","graph_json":"https://pith.science/api/pith-number/VHLLZOWDUYPNU46DDPUWMBK6LN/graph.json","events_json":"https://pith.science/api/pith-number/VHLLZOWDUYPNU46DDPUWMBK6LN/events.json","paper":"https://pith.science/paper/VHLLZOWD"},"agent_actions":{"view_html":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN","download_json":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN.json","view_paper":"https://pith.science/paper/VHLLZOWD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.03381&json=true","fetch_graph":"https://pith.science/api/pith-number/VHLLZOWDUYPNU46DDPUWMBK6LN/graph.json","fetch_events":"https://pith.science/api/pith-number/VHLLZOWDUYPNU46DDPUWMBK6LN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN/action/storage_attestation","attest_author":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN/action/author_attestation","sign_citation":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN/action/citation_signature","submit_replication":"https://pith.science/pith/VHLLZOWDUYPNU46DDPUWMBK6LN/action/replication_record"}},"created_at":"2026-07-05T07:23:56.078637+00:00","updated_at":"2026-07-05T07:23:56.078637+00:00"}