{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Z7QMYGE666LEFEXRTKI43ASQG4","short_pith_number":"pith:Z7QMYGE6","schema_version":"1.0","canonical_sha256":"cfe0cc189ef7964292f19a91cd8250370eee57b426a308e26740a7d12339ae26","source":{"kind":"arxiv","id":"2107.03996","version":3},"attestation_state":"computed","paper":{"title":"Learning Vision-Guided Quadrupedal Locomotion End-to-End with Cross-Modal Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Huazhe Xu, Minghao Zhang, Nicklas Hansen, Ruihan Yang, Xiaolong Wang","submitted_at":"2021-07-08T17:41:55Z","abstract_excerpt":"We propose to address quadrupedal locomotion tasks using Reinforcement Learning (RL) with a Transformer-based model that learns to combine proprioceptive information and high-dimensional depth sensor inputs. While learning-based locomotion has made great advances using RL, most methods still rely on domain randomization for training blind agents that generalize to challenging terrains. Our key insight is that proprioceptive states only offer contact measurements for immediate reaction, whereas an agent equipped with visual sensory observations can learn to proactively maneuver environments wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.03996","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-08T17:41:55Z","cross_cats_sorted":["cs.CV","cs.RO"],"title_canon_sha256":"75486551be3ea33b277a040f9c42acf17397d97d444c3405fc67d6995ae0718c","abstract_canon_sha256":"5327cb7b1bfec01693630f7d76f8fe6797a53217808ff7d92824f841b3a362a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:26:33.121753Z","signature_b64":"vzvjWr+nirGdZLyh4XoiVamvuHep0EoACYkiEcsa/tvfWqOiLTWd2/UAp9FppiFBuVg9VT24k4xPFW9+LYNoAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cfe0cc189ef7964292f19a91cd8250370eee57b426a308e26740a7d12339ae26","last_reissued_at":"2026-07-05T04:26:33.121203Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:26:33.121203Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Vision-Guided Quadrupedal Locomotion End-to-End with Cross-Modal Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Huazhe Xu, Minghao Zhang, Nicklas Hansen, Ruihan Yang, Xiaolong Wang","submitted_at":"2021-07-08T17:41:55Z","abstract_excerpt":"We propose to address quadrupedal locomotion tasks using Reinforcement Learning (RL) with a Transformer-based model that learns to combine proprioceptive information and high-dimensional depth sensor inputs. While learning-based locomotion has made great advances using RL, most methods still rely on domain randomization for training blind agents that generalize to challenging terrains. Our key insight is that proprioceptive states only offer contact measurements for immediate reaction, whereas an agent equipped with visual sensory observations can learn to proactively maneuver environments wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.03996","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.03996/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.03996","created_at":"2026-07-05T04:26:33.121267+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.03996v3","created_at":"2026-07-05T04:26:33.121267+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.03996","created_at":"2026-07-05T04:26:33.121267+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z7QMYGE666LE","created_at":"2026-07-05T04:26:33.121267+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z7QMYGE666LEFEXR","created_at":"2026-07-05T04:26:33.121267+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z7QMYGE6","created_at":"2026-07-05T04:26:33.121267+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25765","citing_title":"StairMaster: Learning to Conquer Risky Hollow Stairs for Agile Quadrupedal Robots","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05873","citing_title":"LadderMan: Learning Humanoid Perceptive Ladder Climbing","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00637","citing_title":"Global-Local Attention Decomposition for Terrain Encoding in Humanoid Perceptive Locomotion","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06382","citing_title":"Now You See That: Learning End-to-End Humanoid Locomotion from Raw Pixels","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4","json":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4.json","graph_json":"https://pith.science/api/pith-number/Z7QMYGE666LEFEXRTKI43ASQG4/graph.json","events_json":"https://pith.science/api/pith-number/Z7QMYGE666LEFEXRTKI43ASQG4/events.json","paper":"https://pith.science/paper/Z7QMYGE6"},"agent_actions":{"view_html":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4","download_json":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4.json","view_paper":"https://pith.science/paper/Z7QMYGE6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.03996&json=true","fetch_graph":"https://pith.science/api/pith-number/Z7QMYGE666LEFEXRTKI43ASQG4/graph.json","fetch_events":"https://pith.science/api/pith-number/Z7QMYGE666LEFEXRTKI43ASQG4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4/action/storage_attestation","attest_author":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4/action/author_attestation","sign_citation":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4/action/citation_signature","submit_replication":"https://pith.science/pith/Z7QMYGE666LEFEXRTKI43ASQG4/action/replication_record"}},"created_at":"2026-07-05T04:26:33.121267+00:00","updated_at":"2026-07-05T04:26:33.121267+00:00"}