{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YFO3WVPTBJJ6JHCX3STU65IDC3","short_pith_number":"pith:YFO3WVPT","schema_version":"1.0","canonical_sha256":"c15dbb55f30a53e49c57dca74f750316f066c001e620ec97f51c66b8168f9c56","source":{"kind":"arxiv","id":"2311.00924","version":1},"attestation_state":"computed","paper":{"title":"The Power of the Senses: Generalizable Manipulation from Vision and Touch through Masked Multimodal Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Carmelo Sferrazza, Hao Liu, Pieter Abbeel, Younggyo Seo, Youngwoon Lee","submitted_at":"2023-11-02T01:33:00Z","abstract_excerpt":"Humans rely on the synergy of their senses for most essential tasks. For tasks requiring object manipulation, we seamlessly and effectively exploit the complementarity of our senses of vision and touch. This paper draws inspiration from such capabilities and aims to find a systematic approach to fuse visual and tactile information in a reinforcement learning setting. We propose Masked Multimodal Learning (M3L), which jointly learns a policy and visual-tactile representations based on masked autoencoding. The representations jointly learned from vision and touch improve sample efficiency, and u"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.00924","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-11-02T01:33:00Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d127cd090ede479813999a6531a421a7752f511e8ccf7f6fcd1d6012ebb0418e","abstract_canon_sha256":"837115facb1345b1c2cc79c4ef7558d41557e999d5b62dedd836876891a26187"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:17.637073Z","signature_b64":"0KAEHAnHEi5fyFykGwqCZR1DijU9HR9/tTn6kCwK7t056kb2T2fTQpGuht7HmCBBf3AopMTcNggK38xPSCdzCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c15dbb55f30a53e49c57dca74f750316f066c001e620ec97f51c66b8168f9c56","last_reissued_at":"2026-07-05T07:08:17.636540Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:17.636540Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Power of the Senses: Generalizable Manipulation from Vision and Touch through Masked Multimodal Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Carmelo Sferrazza, Hao Liu, Pieter Abbeel, Younggyo Seo, Youngwoon Lee","submitted_at":"2023-11-02T01:33:00Z","abstract_excerpt":"Humans rely on the synergy of their senses for most essential tasks. For tasks requiring object manipulation, we seamlessly and effectively exploit the complementarity of our senses of vision and touch. This paper draws inspiration from such capabilities and aims to find a systematic approach to fuse visual and tactile information in a reinforcement learning setting. We propose Masked Multimodal Learning (M3L), which jointly learns a policy and visual-tactile representations based on masked autoencoding. The representations jointly learned from vision and touch improve sample efficiency, and u"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.00924","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.00924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.00924","created_at":"2026-07-05T07:08:17.636613+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.00924v1","created_at":"2026-07-05T07:08:17.636613+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.00924","created_at":"2026-07-05T07:08:17.636613+00:00"},{"alias_kind":"pith_short_12","alias_value":"YFO3WVPTBJJ6","created_at":"2026-07-05T07:08:17.636613+00:00"},{"alias_kind":"pith_short_16","alias_value":"YFO3WVPTBJJ6JHCX","created_at":"2026-07-05T07:08:17.636613+00:00"},{"alias_kind":"pith_short_8","alias_value":"YFO3WVPT","created_at":"2026-07-05T07:08:17.636613+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17055","citing_title":"T-Rex: Tactile-Reactive Dexterous Manipulation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06281","citing_title":"Multi-Resolution Tactile Imitation Learning for Contact-Rich Robotic Manipulation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2504.14820","citing_title":"A Visual Reinforcement Learning-Based Separate Primitive Policy for Peg-in-Hole Tasks","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14427","citing_title":"Self-Supervised Multisensory Pretraining for Contact-Rich Robot Reinforcement Learning","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3","json":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3.json","graph_json":"https://pith.science/api/pith-number/YFO3WVPTBJJ6JHCX3STU65IDC3/graph.json","events_json":"https://pith.science/api/pith-number/YFO3WVPTBJJ6JHCX3STU65IDC3/events.json","paper":"https://pith.science/paper/YFO3WVPT"},"agent_actions":{"view_html":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3","download_json":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3.json","view_paper":"https://pith.science/paper/YFO3WVPT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.00924&json=true","fetch_graph":"https://pith.science/api/pith-number/YFO3WVPTBJJ6JHCX3STU65IDC3/graph.json","fetch_events":"https://pith.science/api/pith-number/YFO3WVPTBJJ6JHCX3STU65IDC3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3/action/storage_attestation","attest_author":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3/action/author_attestation","sign_citation":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3/action/citation_signature","submit_replication":"https://pith.science/pith/YFO3WVPTBJJ6JHCX3STU65IDC3/action/replication_record"}},"created_at":"2026-07-05T07:08:17.636613+00:00","updated_at":"2026-07-05T07:08:17.636613+00:00"}