{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:F7GHSNJF23FGAIW4WUTX3VTUJS","short_pith_number":"pith:F7GHSNJF","schema_version":"1.0","canonical_sha256":"2fcc793525d6ca6022dcb5277dd6744c8fa1c8b468c7471cc6d32a53f41b8ea4","source":{"kind":"arxiv","id":"2008.04460","version":2},"attestation_state":"computed","paper":{"title":"Hardware as Policy: Mechanical and Computational Co-Optimization using Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Matei Ciocarlie, Tianjian Chen, Zhanpeng He","submitted_at":"2020-08-11T00:10:44Z","abstract_excerpt":"Deep Reinforcement Learning (RL) has shown great success in learning complex control policies for a variety of applications in robotics. However, in most such cases, the hardware of the robot has been considered immutable, modeled as part of the environment. In this study, we explore the problem of learning hardware and control parameters together in a unified RL framework. To achieve this, we propose to model the robot body as a \"hardware policy\", analogous to and optimized jointly with its computational counterpart. We show that, by modeling such hardware policies as auto-differentiable comp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2008.04460","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2020-08-11T00:10:44Z","cross_cats_sorted":[],"title_canon_sha256":"21e0fe74d9c6eb79b74ac470b24988b043039f94043b859a480b835d79468504","abstract_canon_sha256":"dd565a0ebc7d91fb303b74930cd4f58ae3984398b7698570c8748932c6d09c89"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:50:11.901290Z","signature_b64":"YvJNKi7u8ouIaNEGU38Rl4yI8t0F4yB1bul2DttVkcu0EcQm/8JGulmFqenhEjipHzhrSM6rShgnbW/NlX+RCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fcc793525d6ca6022dcb5277dd6744c8fa1c8b468c7471cc6d32a53f41b8ea4","last_reissued_at":"2026-07-05T01:50:11.900812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:50:11.900812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hardware as Policy: Mechanical and Computational Co-Optimization using Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Matei Ciocarlie, Tianjian Chen, Zhanpeng He","submitted_at":"2020-08-11T00:10:44Z","abstract_excerpt":"Deep Reinforcement Learning (RL) has shown great success in learning complex control policies for a variety of applications in robotics. However, in most such cases, the hardware of the robot has been considered immutable, modeled as part of the environment. In this study, we explore the problem of learning hardware and control parameters together in a unified RL framework. To achieve this, we propose to model the robot body as a \"hardware policy\", analogous to and optimized jointly with its computational counterpart. We show that, by modeling such hardware policies as auto-differentiable comp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2008.04460","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2008.04460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2008.04460","created_at":"2026-07-05T01:50:11.900865+00:00"},{"alias_kind":"arxiv_version","alias_value":"2008.04460v2","created_at":"2026-07-05T01:50:11.900865+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2008.04460","created_at":"2026-07-05T01:50:11.900865+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7GHSNJF23FG","created_at":"2026-07-05T01:50:11.900865+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7GHSNJF23FGAIW4","created_at":"2026-07-05T01:50:11.900865+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7GHSNJF","created_at":"2026-07-05T01:50:11.900865+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS","json":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS.json","graph_json":"https://pith.science/api/pith-number/F7GHSNJF23FGAIW4WUTX3VTUJS/graph.json","events_json":"https://pith.science/api/pith-number/F7GHSNJF23FGAIW4WUTX3VTUJS/events.json","paper":"https://pith.science/paper/F7GHSNJF"},"agent_actions":{"view_html":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS","download_json":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS.json","view_paper":"https://pith.science/paper/F7GHSNJF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2008.04460&json=true","fetch_graph":"https://pith.science/api/pith-number/F7GHSNJF23FGAIW4WUTX3VTUJS/graph.json","fetch_events":"https://pith.science/api/pith-number/F7GHSNJF23FGAIW4WUTX3VTUJS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS/action/storage_attestation","attest_author":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS/action/author_attestation","sign_citation":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS/action/citation_signature","submit_replication":"https://pith.science/pith/F7GHSNJF23FGAIW4WUTX3VTUJS/action/replication_record"}},"created_at":"2026-07-05T01:50:11.900865+00:00","updated_at":"2026-07-05T01:50:11.900865+00:00"}