{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:V4OTO2IKK4DAPW3JC4EARRO2OV","short_pith_number":"pith:V4OTO2IK","schema_version":"1.0","canonical_sha256":"af1d37690a570607db69170808c5da7554ecf8dfaacd9066af539a5a96543415","source":{"kind":"arxiv","id":"2112.03149","version":1},"attestation_state":"computed","paper":{"title":"Distilled Domain Randomization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Benedikt Hahner, Fabio Muratore, Jan Peters, Julien Brosseit, Michael Gienger","submitted_at":"2021-12-06T16:35:08Z","abstract_excerpt":"Deep reinforcement learning is an effective tool to learn robot control policies from scratch. However, these methods are notorious for the enormous amount of required training data which is prohibitively expensive to collect on real robots. A highly popular alternative is to learn from simulations, allowing to generate the data much faster, safer, and cheaper. Since all simulators are mere models of reality, there are inevitable differences between the simulated and the real data, often referenced as the 'reality gap'. To bridge this gap, many approaches learn one policy from a distribution o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.03149","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-06T16:35:08Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"143d5d0a9609438c883b7e8f3775c36bca6bc9a440ad63c61906eca47be0c683","abstract_canon_sha256":"76f6377c2cc65f06e63c394eb66a126a1d02d6f7047cfd7228b535bb638bdb1a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:38:02.716658Z","signature_b64":"YsGpobySByxgpWjGnqzztoUTd4nK1blQEqxkwI2lDS9lWBTP111K2FG3X3SwzyrS/V8t35OIpKUuSLJ3ecUsAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af1d37690a570607db69170808c5da7554ecf8dfaacd9066af539a5a96543415","last_reissued_at":"2026-07-05T03:38:02.716213Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:38:02.716213Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distilled Domain Randomization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Benedikt Hahner, Fabio Muratore, Jan Peters, Julien Brosseit, Michael Gienger","submitted_at":"2021-12-06T16:35:08Z","abstract_excerpt":"Deep reinforcement learning is an effective tool to learn robot control policies from scratch. However, these methods are notorious for the enormous amount of required training data which is prohibitively expensive to collect on real robots. A highly popular alternative is to learn from simulations, allowing to generate the data much faster, safer, and cheaper. Since all simulators are mere models of reality, there are inevitable differences between the simulated and the real data, often referenced as the 'reality gap'. To bridge this gap, many approaches learn one policy from a distribution o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.03149","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.03149/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.03149","created_at":"2026-07-05T03:38:02.716270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.03149v1","created_at":"2026-07-05T03:38:02.716270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.03149","created_at":"2026-07-05T03:38:02.716270+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4OTO2IKK4DA","created_at":"2026-07-05T03:38:02.716270+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4OTO2IKK4DAPW3J","created_at":"2026-07-05T03:38:02.716270+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4OTO2IK","created_at":"2026-07-05T03:38:02.716270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08054","citing_title":"COMBO-Grasp: Learning Constraint-Based Manipulation for Bimanual Occluded Grasping","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV","json":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV.json","graph_json":"https://pith.science/api/pith-number/V4OTO2IKK4DAPW3JC4EARRO2OV/graph.json","events_json":"https://pith.science/api/pith-number/V4OTO2IKK4DAPW3JC4EARRO2OV/events.json","paper":"https://pith.science/paper/V4OTO2IK"},"agent_actions":{"view_html":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV","download_json":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV.json","view_paper":"https://pith.science/paper/V4OTO2IK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.03149&json=true","fetch_graph":"https://pith.science/api/pith-number/V4OTO2IKK4DAPW3JC4EARRO2OV/graph.json","fetch_events":"https://pith.science/api/pith-number/V4OTO2IKK4DAPW3JC4EARRO2OV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV/action/storage_attestation","attest_author":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV/action/author_attestation","sign_citation":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV/action/citation_signature","submit_replication":"https://pith.science/pith/V4OTO2IKK4DAPW3JC4EARRO2OV/action/replication_record"}},"created_at":"2026-07-05T03:38:02.716270+00:00","updated_at":"2026-07-05T03:38:02.716270+00:00"}