{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BA7U5DNFLQTXCWPM3IDOOO4WXI","short_pith_number":"pith:BA7U5DNF","schema_version":"1.0","canonical_sha256":"083f4e8da55c277159ecda06e73b96ba274e308031ebd320bb513b8829590e2d","source":{"kind":"arxiv","id":"2502.13187","version":3},"attestation_state":"computed","paper":{"title":"A Survey of Sim-to-Real Methods in RL: Progress, Prospects and Challenges with Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Alvaro Velasquez, Hua Wei, Justin Turnau, Longchao Da, Paulo Shakarian, Thirulogasankar Pranav Kutralingam","submitted_at":"2025-02-18T12:57:29Z","abstract_excerpt":"Deep Reinforcement Learning (RL) has been explored and verified to be effective in solving decision-making tasks in various domains, such as robotics, transportation, recommender systems, etc. It learns from the interaction with environments and updates the policy using the collected experience. However, due to the limited real-world data and unbearable consequences of taking detrimental actions, the learning of RL policy is mainly restricted within the simulators. This practice guarantees safety in learning but introduces an inevitable sim-to-real gap in terms of deployment, thus causing degr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.13187","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-18T12:57:29Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"ba0b92de5e0608e34096f9bf17a560bdf2b6cba126afb226ca9e8aefb993bd0c","abstract_canon_sha256":"6d3f8a384e699bb3cbcb27a4454d7c1ca7683f0c3d9682ec4c95679b63192b18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:43.917559Z","signature_b64":"bhSMvU+zn40mcWt1k22SGp4z5xrOu8hUDFxv5luz4P8PziXccnKIZepStd3htxrjfkAQxefoikNpHseB+fHYAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"083f4e8da55c277159ecda06e73b96ba274e308031ebd320bb513b8829590e2d","last_reissued_at":"2026-07-05T10:26:43.917056Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:43.917056Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Sim-to-Real Methods in RL: Progress, Prospects and Challenges with Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Alvaro Velasquez, Hua Wei, Justin Turnau, Longchao Da, Paulo Shakarian, Thirulogasankar Pranav Kutralingam","submitted_at":"2025-02-18T12:57:29Z","abstract_excerpt":"Deep Reinforcement Learning (RL) has been explored and verified to be effective in solving decision-making tasks in various domains, such as robotics, transportation, recommender systems, etc. It learns from the interaction with environments and updates the policy using the collected experience. However, due to the limited real-world data and unbearable consequences of taking detrimental actions, the learning of RL policy is mainly restricted within the simulators. This practice guarantees safety in learning but introduces an inevitable sim-to-real gap in terms of deployment, thus causing degr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13187","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13187/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.13187","created_at":"2026-07-05T10:26:43.917113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.13187v3","created_at":"2026-07-05T10:26:43.917113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13187","created_at":"2026-07-05T10:26:43.917113+00:00"},{"alias_kind":"pith_short_12","alias_value":"BA7U5DNFLQTX","created_at":"2026-07-05T10:26:43.917113+00:00"},{"alias_kind":"pith_short_16","alias_value":"BA7U5DNFLQTXCWPM","created_at":"2026-07-05T10:26:43.917113+00:00"},{"alias_kind":"pith_short_8","alias_value":"BA7U5DNF","created_at":"2026-07-05T10:26:43.917113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26575","citing_title":"IDEA: Insensitive to Dynamics Mismatch via Effect Alignment for Sim-to-Real Transfer in Multi-Agent Control","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12069","citing_title":"Tac-DINO: Learning Vision-Tactile Features with Patch Alignment","ref_index":130,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07017","citing_title":"The Sim-to-Real Gap of Foundation Model Agents: A Unified MDP Perspective","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28476","citing_title":"FADA: Few-Shot Domain Adaptation via Dynamics Alignment for Humanoid Control","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26286","citing_title":"Decoupled Delay Compensation: Enhancing Pre-trained MARL Policies via Learned Dynamics Filtering","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20624","citing_title":"In LLM Reasoning, there is Irrationality on top of Value Misalignment","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09499","citing_title":"Targeting World Models to Compromise Robot Learning Pipelines","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26126","citing_title":"Application of Deep Reinforcement Learning to Event-Triggered Control for Networked Artificial Pancreas Systems","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21282","citing_title":"EgoWalk: A Multimodal Dataset for Robot Navigation in the Wild","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14617","citing_title":"UniCon: A Unified System for Efficient Robot Learning Transfers","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04737","citing_title":"Rationality Measurement and Theory for Reinforcement Learning Agents","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02438","citing_title":"Mitigating Data Scarcity in Spaceflight Applications for Offline Reinforcement Learning Using Physics-Informed Deep Generative Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26126","citing_title":"Application of Deep Reinforcement Learning to Event-Triggered Control for Networked Artificial Pancreas Systems","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02529","citing_title":"Sim-to-Real Transfer and Robustness Evaluation of Reinforcement Learning Control with Integrated Perception on an ASV for Floating Waste Capture","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI","json":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI.json","graph_json":"https://pith.science/api/pith-number/BA7U5DNFLQTXCWPM3IDOOO4WXI/graph.json","events_json":"https://pith.science/api/pith-number/BA7U5DNFLQTXCWPM3IDOOO4WXI/events.json","paper":"https://pith.science/paper/BA7U5DNF"},"agent_actions":{"view_html":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI","download_json":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI.json","view_paper":"https://pith.science/paper/BA7U5DNF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.13187&json=true","fetch_graph":"https://pith.science/api/pith-number/BA7U5DNFLQTXCWPM3IDOOO4WXI/graph.json","fetch_events":"https://pith.science/api/pith-number/BA7U5DNFLQTXCWPM3IDOOO4WXI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI/action/storage_attestation","attest_author":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI/action/author_attestation","sign_citation":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI/action/citation_signature","submit_replication":"https://pith.science/pith/BA7U5DNFLQTXCWPM3IDOOO4WXI/action/replication_record"}},"created_at":"2026-07-05T10:26:43.917113+00:00","updated_at":"2026-07-05T10:26:43.917113+00:00"}