{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PMC47M46CAZZ4YNES3RPH4TEI5","short_pith_number":"pith:PMC47M46","schema_version":"1.0","canonical_sha256":"7b05cfb39e10339e61a496e2f3f264474a060a8bb6d1e4811ff8979762532da6","source":{"kind":"arxiv","id":"2403.03949","version":3},"attestation_state":"computed","paper":{"title":"Reconciling Reality through Simulation: A Real-to-Sim-to-Real Approach for Robust Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Abhishek Gupta, Anthony Simeonov, April Chan, Marcel Torne, Pulkit Agrawal, Tao Chen, Zechu Li","submitted_at":"2024-03-06T18:55:36Z","abstract_excerpt":"Imitation learning methods need significant human supervision to learn policies robust to changes in object poses, physical disturbances, and visual distractors. Reinforcement learning, on the other hand, can explore the environment autonomously to learn robust behaviors but may require impractical amounts of unsafe real-world data collection. To learn performant, robust policies without the burden of unsafe real-world data collection or extensive human supervision, we propose RialTo, a system for robustifying real-world imitation learning policies via reinforcement learning in \"digital twin\" "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.03949","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-03-06T18:55:36Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"815b91a3bcaf7440bf6cb7caedb5f160e6bd3a0d5858ea02857133d211320287","abstract_canon_sha256":"29c6ac8f11ad0becaedc26673c613ae51aacc8f83f69d98309c4fa9b14315593"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:21.592156Z","signature_b64":"nzG1uAPiDcJPGFqbk7K1CiSmXSM/yHPqGvJAOOWDFA7Z2ifRqjmC3keqpDICEPS1W/07hIK7DDTF4JCEsz2zCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b05cfb39e10339e61a496e2f3f264474a060a8bb6d1e4811ff8979762532da6","last_reissued_at":"2026-07-05T09:39:21.591694Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:21.591694Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reconciling Reality through Simulation: A Real-to-Sim-to-Real Approach for Robust Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Abhishek Gupta, Anthony Simeonov, April Chan, Marcel Torne, Pulkit Agrawal, Tao Chen, Zechu Li","submitted_at":"2024-03-06T18:55:36Z","abstract_excerpt":"Imitation learning methods need significant human supervision to learn policies robust to changes in object poses, physical disturbances, and visual distractors. Reinforcement learning, on the other hand, can explore the environment autonomously to learn robust behaviors but may require impractical amounts of unsafe real-world data collection. To learn performant, robust policies without the burden of unsafe real-world data collection or extensive human supervision, we propose RialTo, a system for robustifying real-world imitation learning policies via reinforcement learning in \"digital twin\" "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.03949","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.03949/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.03949","created_at":"2026-07-05T09:39:21.591748+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.03949v3","created_at":"2026-07-05T09:39:21.591748+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.03949","created_at":"2026-07-05T09:39:21.591748+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMC47M46CAZZ","created_at":"2026-07-05T09:39:21.591748+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMC47M46CAZZ4YNE","created_at":"2026-07-05T09:39:21.591748+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMC47M46","created_at":"2026-07-05T09:39:21.591748+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24628","citing_title":"ArtiTwinSplat: Interactable Digital Twin Reconstruction via Gaussian Splatting from RGB-D videos","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22471","citing_title":"Scalable Multi-Task Data Generation via Reinforcement Learning for Language-Conditioned Bimanual Dexterous Manipulation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10366","citing_title":"A Practical Recipe Towards Improving Sim-and-Real Correlation for VLA Evaluation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09813","citing_title":"iMaC: Translating Actions into Motion and Contact Images for Embodied World Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08828","citing_title":"Video2Sim2Real: Full-Stack Autonomous Dexterous Skill Acquisition from a Single Human Video","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08610","citing_title":"HARBOR: A Harness Framework for Agentic Robot Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03949","citing_title":"Preference-Calibrated Human-in-the-Loop Reinforcement Learning for Robotic Manipulation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27475","citing_title":"Support-Constrained RL Enables Real-World Policy Improvement without Real-World Experience","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28276","citing_title":"SimFoundry: Modular and Automated Scene Generation for Policy Learning and Evaluation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22471","citing_title":"Scalable Multi-Task Data Generation via Reinforcement Learning for Language-Conditioned Bimanual Dexterous Manipulation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.16776","citing_title":"JoyAI-Sim: A Simulation-Enabled Interconversion Toolchain for the Embodied Data Pyramid","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09023","citing_title":"TwinRL: Digital Twin-Driven Reinforcement Learning for Real-World Robotic Manipulation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2506.00982","citing_title":"Robust and Safe Multi-Agent Reinforcement Learning with Communication for Autonomous Vehicles: From Simulation to Hardware","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14058","citing_title":"What Matters in Building Vision-Language-Action Models for Generalist Robots","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04415","citing_title":"RoboBPP: Benchmarking Robotic Online Bin Packing with Physics-based Simulation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03066","citing_title":"Redefining End-of-Life: Intelligent Automation for Electronics Remanufacturing Systems","ref_index":205,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27106","citing_title":"Reconstruction by Generation: 3D Multi-Object Scene Reconstruction from Sparse Observations","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05411","citing_title":"Creative Robot Tool Use by Counterfactual Reasoning","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01232","citing_title":"A Principled Approach for Creating High-fidelity Synthetic Demonstrations for Imitation Learning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11386","citing_title":"ComSim: Building Scalable Real-World Robot Data Generation via Compositional Simulation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08544","citing_title":"SIM1: Physics-Aligned Simulator as Zero-Shot Data Scaler in Deformable Worlds","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5","json":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5.json","graph_json":"https://pith.science/api/pith-number/PMC47M46CAZZ4YNES3RPH4TEI5/graph.json","events_json":"https://pith.science/api/pith-number/PMC47M46CAZZ4YNES3RPH4TEI5/events.json","paper":"https://pith.science/paper/PMC47M46"},"agent_actions":{"view_html":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5","download_json":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5.json","view_paper":"https://pith.science/paper/PMC47M46","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.03949&json=true","fetch_graph":"https://pith.science/api/pith-number/PMC47M46CAZZ4YNES3RPH4TEI5/graph.json","fetch_events":"https://pith.science/api/pith-number/PMC47M46CAZZ4YNES3RPH4TEI5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5/action/storage_attestation","attest_author":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5/action/author_attestation","sign_citation":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5/action/citation_signature","submit_replication":"https://pith.science/pith/PMC47M46CAZZ4YNES3RPH4TEI5/action/replication_record"}},"created_at":"2026-07-05T09:39:21.591748+00:00","updated_at":"2026-07-05T09:39:21.591748+00:00"}