{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UZITYVQLTIT6EMUH5URW7MWQMB","short_pith_number":"pith:UZITYVQL","schema_version":"1.0","canonical_sha256":"a6513c560b9a27e23287ed236fb2d0604dd775f88392aee5701d62363980801c","source":{"kind":"arxiv","id":"2410.03076","version":1},"attestation_state":"computed","paper":{"title":"Residual Policy Learning for Perceptive Quadruped Control Using Differentiable Simulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Davide Scaramuzza, Fan Shi, Jing Yuan Luo, Marco Hutter, Victor Klemm, Yunlong Song","submitted_at":"2024-10-04T01:37:54Z","abstract_excerpt":"First-order Policy Gradient (FoPG) algorithms such as Backpropagation through Time and Analytical Policy Gradients leverage local simulation physics to accelerate policy search, significantly improving sample efficiency in robot control compared to standard model-free reinforcement learning. However, FoPG algorithms can exhibit poor learning dynamics in contact-rich tasks like locomotion. Previous approaches address this issue by alleviating contact dynamics via algorithmic or simulation innovations. In contrast, we propose guiding the policy search by learning a residual over a simple baselin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03076","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-10-04T01:37:54Z","cross_cats_sorted":[],"title_canon_sha256":"e02377c96b9e8ed030efa2cdcc9178b4062a413cdfb8975a319768c4a9094577","abstract_canon_sha256":"33f9bbfdd9415e4b4edafd67ee72d6bfa343e0d56e3437d4b7500acdcbc7873f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:43.802354Z","signature_b64":"Zvxt0KhX4dxMXKndoSNYpjKrmq78DhXPPn4mgWEeAufvbvk1NqYxUgNFW81gd+251QoOLo+h7UccFAJkOdRpDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6513c560b9a27e23287ed236fb2d0604dd775f88392aee5701d62363980801c","last_reissued_at":"2026-07-05T09:15:43.801888Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:43.801888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Residual Policy Learning for Perceptive Quadruped Control Using Differentiable Simulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Davide Scaramuzza, Fan Shi, Jing Yuan Luo, Marco Hutter, Victor Klemm, Yunlong Song","submitted_at":"2024-10-04T01:37:54Z","abstract_excerpt":"First-order Policy Gradient (FoPG) algorithms such as Backpropagation through Time and Analytical Policy Gradients leverage local simulation physics to accelerate policy search, significantly improving sample efficiency in robot control compared to standard model-free reinforcement learning. However, FoPG algorithms can exhibit poor learning dynamics in contact-rich tasks like locomotion. Previous approaches address this issue by alleviating contact dynamics via algorithmic or simulation innovations. In contrast, we propose guiding the policy search by learning a residual over a simple baselin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03076","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03076/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03076","created_at":"2026-07-05T09:15:43.801946+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03076v1","created_at":"2026-07-05T09:15:43.801946+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03076","created_at":"2026-07-05T09:15:43.801946+00:00"},{"alias_kind":"pith_short_12","alias_value":"UZITYVQLTIT6","created_at":"2026-07-05T09:15:43.801946+00:00"},{"alias_kind":"pith_short_16","alias_value":"UZITYVQLTIT6EMUH","created_at":"2026-07-05T09:15:43.801946+00:00"},{"alias_kind":"pith_short_8","alias_value":"UZITYVQL","created_at":"2026-07-05T09:15:43.801946+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB","json":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB.json","graph_json":"https://pith.science/api/pith-number/UZITYVQLTIT6EMUH5URW7MWQMB/graph.json","events_json":"https://pith.science/api/pith-number/UZITYVQLTIT6EMUH5URW7MWQMB/events.json","paper":"https://pith.science/paper/UZITYVQL"},"agent_actions":{"view_html":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB","download_json":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB.json","view_paper":"https://pith.science/paper/UZITYVQL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03076&json=true","fetch_graph":"https://pith.science/api/pith-number/UZITYVQLTIT6EMUH5URW7MWQMB/graph.json","fetch_events":"https://pith.science/api/pith-number/UZITYVQLTIT6EMUH5URW7MWQMB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB/action/storage_attestation","attest_author":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB/action/author_attestation","sign_citation":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB/action/citation_signature","submit_replication":"https://pith.science/pith/UZITYVQLTIT6EMUH5URW7MWQMB/action/replication_record"}},"created_at":"2026-07-05T09:15:43.801946+00:00","updated_at":"2026-07-05T09:15:43.801946+00:00"}