{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:7HPQQIFWD3BEYX4D7FDFZ7NFCW","short_pith_number":"pith:7HPQQIFW","schema_version":"1.0","canonical_sha256":"f9df0820b61ec24c5f83f9465cfda51589e7bf34952d5c7a199c5ead2bc064a1","source":{"kind":"arxiv","id":"2110.03239","version":2},"attestation_state":"computed","paper":{"title":"Understanding Domain Randomization for Sim-to-real Transfer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chi Jin, Jiachen Hu, Lihong Li, Liwei Wang, Xiaoyu Chen","submitted_at":"2021-10-07T07:45:59Z","abstract_excerpt":"Reinforcement learning encounters many challenges when applied directly in the real world. Sim-to-real transfer is widely used to transfer the knowledge learned from simulation to the real world. Domain randomization -- one of the most popular algorithms for sim-to-real transfer -- has been demonstrated to be effective in various tasks in robotics and autonomous driving. Despite its empirical successes, theoretical understanding on why this simple algorithm works is limited. In this paper, we propose a theoretical framework for sim-to-real transfers, in which the simulator is modeled as a set "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.03239","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-10-07T07:45:59Z","cross_cats_sorted":[],"title_canon_sha256":"ec480338a098c6b9abbecd5e14b760d31cdaab5a7378118e0b08588b450c2d73","abstract_canon_sha256":"8e6777be6d52723fd37754c7aa560c3fdb1b5b2f35ed8f18d10ab14272b0bb63"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:42.049808Z","signature_b64":"pCHEJyM/T4Fdpcd1/RgQX0Wla73SaRKUDAWK6IeTPfFvuWpMS009KJvK0IRbgEJUlKHS4zLkzxd535rKKjxOBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9df0820b61ec24c5f83f9465cfda51589e7bf34952d5c7a199c5ead2bc064a1","last_reissued_at":"2026-07-05T04:04:42.049402Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:42.049402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Domain Randomization for Sim-to-real Transfer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chi Jin, Jiachen Hu, Lihong Li, Liwei Wang, Xiaoyu Chen","submitted_at":"2021-10-07T07:45:59Z","abstract_excerpt":"Reinforcement learning encounters many challenges when applied directly in the real world. Sim-to-real transfer is widely used to transfer the knowledge learned from simulation to the real world. Domain randomization -- one of the most popular algorithms for sim-to-real transfer -- has been demonstrated to be effective in various tasks in robotics and autonomous driving. Despite its empirical successes, theoretical understanding on why this simple algorithm works is limited. In this paper, we propose a theoretical framework for sim-to-real transfers, in which the simulator is modeled as a set "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.03239","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.03239/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.03239","created_at":"2026-07-05T04:04:42.049457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.03239v2","created_at":"2026-07-05T04:04:42.049457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.03239","created_at":"2026-07-05T04:04:42.049457+00:00"},{"alias_kind":"pith_short_12","alias_value":"7HPQQIFWD3BE","created_at":"2026-07-05T04:04:42.049457+00:00"},{"alias_kind":"pith_short_16","alias_value":"7HPQQIFWD3BEYX4D","created_at":"2026-07-05T04:04:42.049457+00:00"},{"alias_kind":"pith_short_8","alias_value":"7HPQQIFW","created_at":"2026-07-05T04:04:42.049457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22579","citing_title":"Stationary Robust Mean-Field Games under Model Mismatches","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22145","citing_title":"Zero-shot Transfer of Reinforcement Learning Control Policies for the Swing-Up and Stabilization of a Cart-Pole System","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09183","citing_title":"Learning When to Stop: Selective Imitation Learning Under Arbitrary Dynamics Shift","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06382","citing_title":"Now You See That: Learning End-to-End Humanoid Locomotion from Raw Pixels","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09183","citing_title":"Learning When to Stop: Selective Imitation Learning Under Arbitrary Dynamics Shift","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW","json":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW.json","graph_json":"https://pith.science/api/pith-number/7HPQQIFWD3BEYX4D7FDFZ7NFCW/graph.json","events_json":"https://pith.science/api/pith-number/7HPQQIFWD3BEYX4D7FDFZ7NFCW/events.json","paper":"https://pith.science/paper/7HPQQIFW"},"agent_actions":{"view_html":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW","download_json":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW.json","view_paper":"https://pith.science/paper/7HPQQIFW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.03239&json=true","fetch_graph":"https://pith.science/api/pith-number/7HPQQIFWD3BEYX4D7FDFZ7NFCW/graph.json","fetch_events":"https://pith.science/api/pith-number/7HPQQIFWD3BEYX4D7FDFZ7NFCW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW/action/storage_attestation","attest_author":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW/action/author_attestation","sign_citation":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW/action/citation_signature","submit_replication":"https://pith.science/pith/7HPQQIFWD3BEYX4D7FDFZ7NFCW/action/replication_record"}},"created_at":"2026-07-05T04:04:42.049457+00:00","updated_at":"2026-07-05T04:04:42.049457+00:00"}